{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/discriminative-fine-tuning/papers/16","list_of":"/method/discriminative-fine-tuning","method":"Discriminative Fine-Tuning","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":16,"pages_in_order":20,"rows_per_page":100,"rows":[1501,1600],"of":1990,"counts":{"archive_papers_tagged":1990,"with_a_code_link":794,"where_syntology_ran_a_sample":271,"not_listed_spam_title":0,"listed":1990,"listed_where_code_ran":271,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":223,"every_run_a_failure_of_syntologys_instrument":48,"listed_with_a_run_with_no_instrument_failure":223,"listed_every_run_a_failure_of_syntologys_instrument":48,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/discriminative-fine-tuning","prev":"/method/discriminative-fine-tuning/papers/15","next":"/method/discriminative-fine-tuning/papers/17","papers":[{"paper":"/paper/predictive-querying-for-autoregressive-neural","slug":"predictive-querying-for-autoregressive-neural","title":"Predictive Querying for Autoregressive Neural Sequence Models","date":"2022-10-12","arxiv_id":"2210.06464","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":2,"n_instrument":5,"unverified":6,"pointer_only":0,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["ajboyd2/prob_seq_queries"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sumbot-summarizing-context-in-open-domain","title":"SUMBot: Summarizing Context in Open-Domain Dialogue Systems","date":"2022-10-12","arxiv_id":"2210.06496","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-pre-trained-transformers-into","slug":"fine-tuning-pre-trained-transformers-into","title":"Fine-Tuning Pre-trained Transformers into Decaying Fast Weights","date":"2022-10-09","arxiv_id":"2210.04243","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jenni-ai/t2fw"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"alphatuning-quantization-aware-parameter","title":"AlphaTuning: Quantization-Aware Parameter-Efficient Adaptation of Large-Scale Pre-Trained Language Models","date":"2022-10-08","arxiv_id":"2210.03858","n_code_links":0,"syntology":null},{"paper":"/paper/exploiting-selection-bias-on-underspecified","slug":"exploiting-selection-bias-on-underspecified","title":"Underspecification in Language Modeling Tasks: A Causality-Informed Study of Gendered Pronoun Resolution","date":"2022-09-30","arxiv_id":"2210.00131","n_code_links":2,"syntology":null},{"paper":"/paper/smallcap-lightweight-image-captioning","slug":"smallcap-lightweight-image-captioning","title":"SmallCap: Lightweight Image Captioning Prompted with Retrieval Augmentation","date":"2022-09-30","arxiv_id":"2209.15323","n_code_links":1,"syntology":null},{"paper":null,"slug":"dfx-a-low-latency-multi-fpga-appliance-for","title":"DFX: A Low-latency Multi-FPGA Appliance for Accelerating Transformer-based Text Generation","date":"2022-09-22","arxiv_id":"2209.10797","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-revealer-private-text-reconstruction-via","title":"Text Revealer: Private Text Reconstruction via Model Inversion Attacks against Transformers","date":"2022-09-21","arxiv_id":"2209.10505","n_code_links":0,"syntology":null},{"paper":null,"slug":"chain-of-explanation-new-prompting-method-to","title":"Chain of Explanation: New Prompting Method to Generate Higher Quality Natural Language Explanation for Implicit Hate Speech","date":"2022-09-11","arxiv_id":"2209.04889","n_code_links":0,"syntology":null},{"paper":null,"slug":"why-so-toxic-measuring-and-triggering-toxic","title":"Why So Toxic? Measuring and Triggering Toxic Behavior in Open-Domain Chatbots","date":"2022-09-07","arxiv_id":"2209.03463","n_code_links":0,"syntology":null},{"paper":null,"slug":"every-picture-tells-a-story-image-grounded","title":"Every picture tells a story: Image-grounded controllable stylistic story generation","date":"2022-09-04","arxiv_id":"2209.01638","n_code_links":0,"syntology":null},{"paper":"/paper/using-large-language-models-to-simulate","slug":"using-large-language-models-to-simulate","title":"Using Large Language Models to Simulate Multiple Humans and Replicate Human Subject Studies","date":"2022-08-18","arxiv_id":"2208.10264","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["gatiaher/using-large-language-models-to-replicate-human-subject-studies"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/neural-embeddings-for-text","slug":"neural-embeddings-for-text","title":"Neural Embeddings for Text","date":"2022-08-17","arxiv_id":"2208.08386","n_code_links":1,"syntology":null},{"paper":"/paper/mocapact-a-multi-task-dataset-for-simulated","slug":"mocapact-a-multi-task-dataset-for-simulated","title":"MoCapAct: A Multi-Task Dataset for Simulated Humanoid Control","date":"2022-08-15","arxiv_id":"2208.07363","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/MoCapAct"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/adan-adaptive-nesterov-momentum-algorithm-for","slug":"adan-adaptive-nesterov-momentum-algorithm-for","title":"Adan: Adaptive Nesterov Momentum Algorithm for Faster Optimizing Deep Models","date":"2022-08-13","arxiv_id":"2208.06677","n_code_links":9,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/adan"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"studying-writer-suggestion-interaction-a","title":"Interacting with next-phrase suggestions: How suggestion systems aid and influence the cognitive processes of writing","date":"2022-08-01","arxiv_id":"2208.00636","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-video-captioning-with-evolving","slug":"zero-shot-video-captioning-with-evolving","title":"Zero-Shot Video Captioning with Evolving Pseudo-Tokens","date":"2022-07-22","arxiv_id":"2207.11100","n_code_links":1,"syntology":null},{"paper":null,"slug":"word-play-for-playing-othello-reverses","title":"Word Play for Playing Othello (Reverses)","date":"2022-07-18","arxiv_id":"2207.08766","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-data-pattern-extraction-attacks-on","title":"Combing for Credentials: Active Pattern Extraction from Smart Reply","date":"2022-07-14","arxiv_id":"2207.10802","n_code_links":0,"syntology":null},{"paper":null,"slug":"few-shot-training-llms-for-project-specific","title":"Few-shot training LLMs for project-specific code-summarization","date":"2022-07-09","arxiv_id":"2207.04237","n_code_links":0,"syntology":null},{"paper":null,"slug":"hidden-schema-networks","title":"Hidden Schema Networks","date":"2022-07-08","arxiv_id":"2207.03777","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-language-models-are-not-born-equal-to","title":"Neural Language Models are not Born Equal to Fit Brain Data, but Training Helps","date":"2022-07-07","arxiv_id":"2207.03380","n_code_links":0,"syntology":null},{"paper":null,"slug":"sensitivity-analysis-on-transferred-neural","title":"Sensitivity Analysis on Transferred Neural Architectures of BERT and GPT-2 for Financial Sentiment Analysis","date":"2022-07-07","arxiv_id":"2207.03037","n_code_links":0,"syntology":null},{"paper":"/paper/materials-transformers-language-models-for","slug":"materials-transformers-language-models-for","title":"Materials Transformers Language Models for Generative Materials Design: a benchmark study","date":"2022-06-27","arxiv_id":"2206.13578","n_code_links":1,"syntology":null},{"paper":null,"slug":"explainable-and-high-performance-hate-and","title":"Explainable and High-Performance Hate and Offensive Speech Detection","date":"2022-06-26","arxiv_id":"2206.12983","n_code_links":0,"syntology":null},{"paper":null,"slug":"cocopie-xgen-a-full-stack-ai-oriented","title":"CoCoPIE XGen: A Full-Stack AI-Oriented Optimizing Framework","date":"2022-06-21","arxiv_id":"2206.10620","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-the-imitation-game-quantifying-and","slug":"beyond-the-imitation-game-quantifying-and","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","date":"2022-06-09","arxiv_id":"2206.04615","n_code_links":6,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google/BIG-bench"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"differentially-private-model-compression","title":"Differentially Private Model Compression","date":"2022-06-03","arxiv_id":"2206.01838","n_code_links":0,"syntology":null},{"paper":null,"slug":"romantic-computing","title":"Romantic-Computing","date":"2022-06-01","arxiv_id":"2206.11864","n_code_links":0,"syntology":null},{"paper":"/paper/multi-agent-reinforcement-learning-is-a","slug":"multi-agent-reinforcement-learning-is-a","title":"Multi-Agent Reinforcement Learning is a Sequence Modeling Problem","date":"2022-05-30","arxiv_id":"2205.14953","n_code_links":1,"syntology":null},{"paper":"/paper/cped-a-large-scale-chinese-personalized-and-1","slug":"cped-a-large-scale-chinese-personalized-and-1","title":"CPED: A Large-Scale Chinese Personalized and Emotional Dialogue Dataset for Conversational AI","date":"2022-05-29","arxiv_id":"2205.14727","n_code_links":1,"syntology":null},{"paper":"/paper/flashattention-fast-and-memory-efficient","slug":"flashattention-fast-and-memory-efficient","title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","date":"2022-05-27","arxiv_id":"2205.14135","n_code_links":13,"syntology":{"ran":24,"of":30,"n_ran_checked":18,"n_instrument":6,"unverified":6,"pointer_only":1,"phrase":"24 ran (of which 2 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","official":{"repos":["dao-ailab/flash-attention"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"paper":"/paper/nearest-neighbor-zero-shot-inference","slug":"nearest-neighbor-zero-shot-inference","title":"kNN-Prompt: Nearest Neighbor Zero-Shot Inference","date":"2022-05-27","arxiv_id":"2205.13792","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-understanding-label-regularization","title":"Do we need Label Regularization to Fine-tune Pre-trained Language Models?","date":"2022-05-25","arxiv_id":"2205.12428","n_code_links":0,"syntology":null},{"paper":"/paper/transcormer-transformer-for-sentence-scoring","slug":"transcormer-transformer-for-sentence-scoring","title":"Transcormer: Transformer for Sentence Scoring with Sliding Language Modeling","date":"2022-05-25","arxiv_id":"2205.12986","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/garden-path-traversal-within-gpt-2","slug":"garden-path-traversal-within-gpt-2","title":"Garden-Path Traversal in GPT-2","date":"2022-05-24","arxiv_id":"2205.12302","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wjurayj/garden-path-gpt2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-the-role-of-bidirectionality-in-language","title":"On the Role of Bidirectionality in Language Model Pre-Training","date":"2022-05-24","arxiv_id":"2205.11726","n_code_links":0,"syntology":null},{"paper":"/paper/graphmae-self-supervised-masked-graph","slug":"graphmae-self-supervised-masked-graph","title":"GraphMAE: Self-Supervised Masked Graph Autoencoders","date":"2022-05-22","arxiv_id":"2205.10803","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thudm/graphmae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/life-after-bert-what-do-other-muppets-2","slug":"life-after-bert-what-do-other-muppets-2","title":"Life after BERT: What do Other Muppets Understand about Language?","date":"2022-05-21","arxiv_id":"2205.10696","n_code_links":1,"syntology":null},{"paper":"/paper/prototypical-calibration-for-few-shot","slug":"prototypical-calibration-for-few-shot","title":"Prototypical Calibration for Few-shot Learning of Language Models","date":"2022-05-20","arxiv_id":"2205.10183","n_code_links":1,"syntology":null},{"paper":"/paper/automated-scoring-for-reading-comprehension","slug":"automated-scoring-for-reading-comprehension","title":"Automated Scoring for Reading Comprehension via In-context BERT Tuning","date":"2022-05-19","arxiv_id":"2205.09864","n_code_links":1,"syntology":null},{"paper":"/paper/towards-understanding-gender-seniority","slug":"towards-understanding-gender-seniority","title":"Towards Understanding Gender-Seniority Compound Bias in Natural Language Generation","date":"2022-05-19","arxiv_id":"2205.09830","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluation-of-transfer-learning-for-polish-1","title":"Evaluation of Transfer Learning for Polish with a Text-to-Text Model","date":"2022-05-18","arxiv_id":"2205.08808","n_code_links":0,"syntology":null},{"paper":"/paper/what-gpt-knows-about-who-is-who-1","slug":"what-gpt-knows-about-who-is-who-1","title":"What GPT Knows About Who is Who","date":"2022-05-16","arxiv_id":"2205.07407","n_code_links":1,"syntology":null},{"paper":"/paper/naturalistic-causal-probing-for-morpho-syntax","slug":"naturalistic-causal-probing-for-morpho-syntax","title":"Naturalistic Causal Probing for Morpho-Syntax","date":"2022-05-14","arxiv_id":"2205.07043","n_code_links":1,"syntology":null},{"paper":null,"slug":"ratatouille-a-tool-for-novel-recipe","title":"Ratatouille: A tool for Novel Recipe Generation","date":"2022-05-10","arxiv_id":"2206.08267","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-segment-preserving-sampling-for-deep","title":"Multi-segment preserving sampling for deep manifold sampler","date":"2022-05-09","arxiv_id":"2205.04259","n_code_links":0,"syntology":null},{"paper":"/paper/when-a-sentence-does-not-introduce-a-1","slug":"when-a-sentence-does-not-introduce-a-1","title":"When a sentence does not introduce a discourse entity, Transformer-based models still sometimes refer to it","date":"2022-05-06","arxiv_id":"2205.03472","n_code_links":1,"syntology":null},{"paper":"/paper/provably-confidential-language-modelling-1","slug":"provably-confidential-language-modelling-1","title":"Provably Confidential Language Modelling","date":"2022-05-04","arxiv_id":"2205.01863","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xuandongzhao/crt"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/hiner-a-large-hindi-named-entity-recognition","slug":"hiner-a-large-hindi-named-entity-recognition","title":"HiNER: A Large Hindi Named Entity Recognition Dataset","date":"2022-04-28","arxiv_id":"2204.13743","n_code_links":1,"syntology":null},{"paper":null,"slug":"tailor-a-prompt-based-approach-to-attribute","title":"Tailor: A Prompt-Based Approach to Attribute-Based Controlled Text Generation","date":"2022-04-28","arxiv_id":"2204.13362","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-artificial-intelligence-as-a","title":"Measuring artificial intelligence: a systematic assessment and implications for governance","date":"2022-04-21","arxiv_id":"2204.10304","n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-tokenization-on-language-models-an-1","title":"Impact of Tokenization on Language Models: An Analysis for Turkish","date":"2022-04-19","arxiv_id":"2204.08832","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-entity-and-tweet-characterization","title":"Zero-shot Entity and Tweet Characterization with Designed Conditional Prompts and Contexts","date":"2022-04-18","arxiv_id":"2204.08405","n_code_links":0,"syntology":null},{"paper":"/paper/mgpt-few-shot-learners-go-multilingual","slug":"mgpt-few-shot-learners-go-multilingual","title":"mGPT: Few-Shot Learners Go Multilingual","date":"2022-04-15","arxiv_id":"2204.07580","n_code_links":1,"syntology":null},{"paper":"/paper/polling-latent-opinions-a-method-for-1","slug":"polling-latent-opinions-a-method-for-1","title":"Polling Latent Opinions: A Method for Computational Sociolinguistics Using Transformer Language Models","date":"2022-04-15","arxiv_id":"2204.07483","n_code_links":1,"syntology":null},{"paper":null,"slug":"brazilian-court-documents-clustered-by","title":"Analysing similarities between legal court documents using natural language processing approaches based on Transformers","date":"2022-04-14","arxiv_id":"2204.07182","n_code_links":0,"syntology":null},{"paper":"/paper/uniform-complexity-for-text-generation","slug":"uniform-complexity-for-text-generation","title":"Uniform Complexity for Text Generation","date":"2022-04-11","arxiv_id":"2204.05185","n_code_links":1,"syntology":null},{"paper":null,"slug":"foundationlayernorm-scaling-bert-and-gpt-to","title":"FoundationLayerNorm: Scaling BERT and GPT to 1,000 Layers","date":"2022-04-09","arxiv_id":"2204.04477","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-attention-through-gradient-based","title":"Accelerating Attention through Gradient-Based Learned Runtime Pruning","date":"2022-04-07","arxiv_id":"2204.03227","n_code_links":0,"syntology":null},{"paper":"/paper/testing-the-limits-of-natural-language-models","slug":"testing-the-limits-of-natural-language-models","title":"Testing the limits of natural language models for predicting human language judgments","date":"2022-04-07","arxiv_id":"2204.03592","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-infused-decoding-1","slug":"knowledge-infused-decoding-1","title":"Knowledge Infused Decoding","date":"2022-04-06","arxiv_id":"2204.03084","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/kid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/data-augmentation-for-intent-classification-1","slug":"data-augmentation-for-intent-classification-1","title":"Data Augmentation for Intent Classification with Off-the-shelf Large Language Models","date":"2022-04-05","arxiv_id":"2204.01959","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["elementai/data-augmentation-with-llms"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"effect-and-analysis-of-large-scale-language","title":"Effect and Analysis of Large-scale Language Model Rescoring on Competitive ASR Systems","date":"2022-04-01","arxiv_id":"2204.00212","n_code_links":0,"syntology":null},{"paper":"/paper/monarch-expressive-structured-matrices-for","slug":"monarch-expressive-structured-matrices-for","title":"Monarch: Expressive Structured Matrices for Efficient and Accurate Training","date":"2022-04-01","arxiv_id":"2204.00595","n_code_links":2,"syntology":{"ran":15,"of":15,"n_ran_checked":14,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hazyresearch/monarch"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"create-a-benchmark-for-chinese-short-video-1","title":"CREATE: A Benchmark for Chinese Short Video Retrieval and Title Generation","date":"2022-03-31","arxiv_id":"2203.16763","n_code_links":0,"syntology":null},{"paper":"/paper/bailando-3d-dance-generation-by-actor-critic","slug":"bailando-3d-dance-generation-by-actor-critic","title":"Bailando: 3D Dance Generation by Actor-Critic GPT with Choreographic Memory","date":"2022-03-24","arxiv_id":"2203.13055","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lisiyao21/bailando"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-supervision-through-random-segments-with","title":"Self-supervision through Random Segments with Autoregressive Coding (RandSAC)","date":"2022-03-22","arxiv_id":"2203.12054","n_code_links":0,"syntology":null},{"paper":"/paper/a-slot-is-not-built-in-one-utterance-spoken-1","slug":"a-slot-is-not-built-in-one-utterance-spoken-1","title":"A Slot Is Not Built in One Utterance: Spoken Language Dialogs with Sub-Slots","date":"2022-03-21","arxiv_id":"2203.10759","n_code_links":1,"syntology":null},{"paper":null,"slug":"compression-of-generative-pre-trained","title":"Compression of Generative Pre-trained Language Models via Quantization","date":"2022-03-21","arxiv_id":"2203.10705","n_code_links":0,"syntology":null},{"paper":"/paper/dependency-based-mixture-language-models","slug":"dependency-based-mixture-language-models","title":"Dependency-based Mixture Language Models","date":"2022-03-19","arxiv_id":"2203.10256","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fadedcosine/dependency-guided-neural-text-generation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"are-you-robert-or-roberta-deceiving-online","title":"Are You Robert or RoBERTa? Deceiving Online Authorship Attribution Models Using Neural Text Generators","date":"2022-03-18","arxiv_id":"2203.09813","n_code_links":0,"syntology":null},{"paper":"/paper/do-language-models-plagiarize","slug":"do-language-models-plagiarize","title":"Do Language Models Plagiarize?","date":"2022-03-15","arxiv_id":"2203.07618","n_code_links":1,"syntology":null},{"paper":null,"slug":"contrastive-visual-semantic-pretraining","title":"Contrastive Visual Semantic Pretraining Magnifies the Semantics of Natural Language Representations","date":"2022-03-14","arxiv_id":"2203.07511","n_code_links":0,"syntology":null},{"paper":"/paper/grips-gradient-free-edit-based-instruction","slug":"grips-gradient-free-edit-based-instruction","title":"GrIPS: Gradient-free, Edit-based Instruction Search for Prompting Large Language Models","date":"2022-03-14","arxiv_id":"2203.07281","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["archiki/grips"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/vast-the-valence-assessing-semantics-test-for","slug":"vast-the-valence-assessing-semantics-test-for","title":"VAST: The Valence-Assessing Semantics Test for Contextualizing Language Models","date":"2022-03-14","arxiv_id":"2203.07504","n_code_links":1,"syntology":null},{"paper":"/paper/elle-efficient-lifelong-pre-training-for-1","slug":"elle-efficient-lifelong-pre-training-for-1","title":"ELLE: Efficient Lifelong Pre-training for Emerging Data","date":"2022-03-12","arxiv_id":"2203.06311","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":1,"n_instrument":3,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thunlp/elle"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/block-sparse-adversarial-attack-to-fool","slug":"block-sparse-adversarial-attack-to-fool","title":"Block-Sparse Adversarial Attack to Fool Transformer-Based Text Classifiers","date":"2022-03-11","arxiv_id":"2203.05948","n_code_links":1,"syntology":null},{"paper":"/paper/when-classifying-grammatical-role-bert-doesn-1","slug":"when-classifying-grammatical-role-bert-doesn-1","title":"When classifying grammatical role, BERT doesn't care about word order... except when it matters","date":"2022-03-11","arxiv_id":"2203.06204","n_code_links":1,"syntology":null},{"paper":"/paper/nlx-gpt-a-model-for-natural-language","slug":"nlx-gpt-a-model-for-natural-language","title":"NLX-GPT: A Model for Natural Language Explanations in Vision and Vision-Language Tasks","date":"2022-03-09","arxiv_id":"2203.05081","n_code_links":1,"syntology":null},{"paper":"/paper/litetransformersearch-training-free-on-device","slug":"litetransformersearch-training-free-on-device","title":"LiteTransformerSearch: Training-free Neural Architecture Search for Efficient Language Models","date":"2022-03-04","arxiv_id":"2203.02094","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/archai"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/parameter-efficient-mixture-of-experts","slug":"parameter-efficient-mixture-of-experts","title":"Parameter-Efficient Mixture-of-Experts Architecture for Pre-trained Language Models","date":"2022-03-02","arxiv_id":"2203.01104","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rucaibox/mpo","rucaibox/mpoe"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-and-adapting-chinese-gpt-to-pinyin-1","slug":"exploring-and-adapting-chinese-gpt-to-pinyin-1","title":"Exploring and Adapting Chinese GPT to Pinyin Input Method","date":"2022-03-01","arxiv_id":"2203.00249","n_code_links":1,"syntology":null},{"paper":"/paper/a-systematic-evaluation-of-large-language","slug":"a-systematic-evaluation-of-large-language","title":"A Systematic Evaluation of Large Language Models of Code","date":"2022-02-26","arxiv_id":"2202.13169","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eleutherai/gpt-neox","vhellendoorn/code-lms"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"consistent-dropout-for-policy-gradient","title":"Consistent Dropout for Policy Gradient Reinforcement Learning","date":"2022-02-23","arxiv_id":"2202.11818","n_code_links":0,"syntology":null},{"paper":"/paper/sgpt-gpt-sentence-embeddings-for-semantic","slug":"sgpt-gpt-sentence-embeddings-for-semantic","title":"SGPT: GPT Sentence Embeddings for Semantic Search","date":"2022-02-17","arxiv_id":"2202.08904","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["muennighoff/sgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"defending-against-reconstruction-attacks-with","title":"Defending against Reconstruction Attacks with Rényi Differential Privacy","date":"2022-02-15","arxiv_id":"2202.07623","n_code_links":0,"syntology":null},{"paper":"/paper/maximizing-communication-efficiency-for-large","slug":"maximizing-communication-efficiency-for-large","title":"Maximizing Communication Efficiency for Large-scale Training via 0/1 Adam","date":"2022-02-12","arxiv_id":"2202.06009","n_code_links":1,"syntology":null},{"paper":"/paper/what-are-the-best-systems-new-perspectives-on","slug":"what-are-the-best-systems-new-perspectives-on","title":"What are the best systems? New perspectives on NLP Benchmarking","date":"2022-02-08","arxiv_id":"2202.03799","n_code_links":1,"syntology":null},{"paper":"/paper/a-benchmark-corpus-for-the-detection-of","slug":"a-benchmark-corpus-for-the-detection-of","title":"A Benchmark Corpus for the Detection of Automatically Generated Text in Academic Publications","date":"2022-02-04","arxiv_id":"2202.02013","n_code_links":1,"syntology":null},{"paper":"/paper/l3cube-mahacorpus-and-mahabert-marathi","slug":"l3cube-mahacorpus-and-mahabert-marathi","title":"L3Cube-MahaCorpus and MahaBERT: Marathi Monolingual Corpus, Marathi BERT Language Models, and Resources","date":"2022-02-02","arxiv_id":"2202.01159","n_code_links":1,"syntology":null},{"paper":null,"slug":"vc-gpt-visual-conditioned-gpt-for-end-to-end","title":"A Frustratingly Simple Approach for End-to-End Image Captioning","date":"2022-01-30","arxiv_id":"2201.12723","n_code_links":0,"syntology":null},{"paper":null,"slug":"dnnfuser-generative-pre-trained-transformer","title":"DNNFuser: Generative Pre-Trained Transformer as a Generalized Mapper for Layer Fusion in DNN Accelerators","date":"2022-01-26","arxiv_id":"2201.11218","n_code_links":0,"syntology":null},{"paper":null,"slug":"pre-trained-language-transformers-are","title":"Pre-Trained Language Transformers are Universal Image Classifiers","date":"2022-01-25","arxiv_id":"2201.10182","n_code_links":0,"syntology":null},{"paper":null,"slug":"synthetic-books","title":"Synthetic Books","date":"2022-01-24","arxiv_id":"2201.09518","n_code_links":0,"syntology":null},{"paper":null,"slug":"auto-regressive-text-generation-with-pre","title":"Auto-regressive Text Generation with Pre-Trained Language Models: An Empirical Study on Question-type Short Text Generation","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-hierarchical-domain-adaptation-for-1","title":"Efficient Hierarchical Domain Adaptation for Pretrained Language Models","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"elastic-weight-consolidation-for-reduction-of","title":"Elastic Weight Consolidation for Reduction of Catastrophic Forgetting in GPT-2","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"jointly-reinforced-user-simulator-and-task","title":"Jointly Reinforced User Simulator and Task-oriented Dialog System with Simplified Generative Architecture","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"polling-latent-opinions-a-method-for","title":"Polling Latent Opinions: A Method for Computational Sociolinguistics Using Transformer Language Models","date":"2022-01-16","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"f01083a8ea407d7b5002447f8e4687118c497c73fbb169b63dc8e80d7f7f0d53","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}