{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/cosine-annealing/papers/33","list_of":"/method/cosine-annealing","method":"Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":33,"pages_in_order":40,"rows_per_page":100,"rows":[3201,3300],"of":3965,"counts":{"archive_papers_tagged":3965,"with_a_code_link":1734,"where_syntology_ran_a_sample":627,"not_listed_spam_title":0,"listed":3965,"listed_where_code_ran":627,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":513,"every_run_a_failure_of_syntologys_instrument":114,"listed_with_a_run_with_no_instrument_failure":513,"listed_every_run_a_failure_of_syntologys_instrument":114,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/cosine-annealing","prev":"/method/cosine-annealing/papers/32","next":"/method/cosine-annealing/papers/34","papers":[{"paper":null,"slug":"romantic-computing","title":"Romantic-Computing","date":"2022-06-01","arxiv_id":"2206.11864","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-deep-learning-a-case-study-in","title":"Knowledge Graph - Deep Learning: A Case Study in Question Answering in Aviation Safety Domain","date":"2022-05-31","arxiv_id":"2205.15952","n_code_links":0,"syntology":null},{"paper":"/paper/billions-of-parameters-are-worth-more-than-in","slug":"billions-of-parameters-are-worth-more-than-in","title":"Billions of Parameters Are Worth More Than In-domain Training Data: A case study in the Legal Case Entailment Task","date":"2022-05-30","arxiv_id":"2205.15172","n_code_links":1,"syntology":null},{"paper":"/paper/multi-agent-reinforcement-learning-is-a","slug":"multi-agent-reinforcement-learning-is-a","title":"Multi-Agent Reinforcement Learning is a Sequence Modeling Problem","date":"2022-05-30","arxiv_id":"2205.14953","n_code_links":1,"syntology":null},{"paper":"/paper/cped-a-large-scale-chinese-personalized-and-1","slug":"cped-a-large-scale-chinese-personalized-and-1","title":"CPED: A Large-Scale Chinese Personalized and Emotional Dialogue Dataset for Conversational AI","date":"2022-05-29","arxiv_id":"2205.14727","n_code_links":1,"syntology":null},{"paper":"/paper/teaching-models-to-express-their-uncertainty","slug":"teaching-models-to-express-their-uncertainty","title":"Teaching Models to Express Their Uncertainty in Words","date":"2022-05-28","arxiv_id":"2205.14334","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sylinrl/calibratedmath"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/flashattention-fast-and-memory-efficient","slug":"flashattention-fast-and-memory-efficient","title":"FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness","date":"2022-05-27","arxiv_id":"2205.14135","n_code_links":13,"syntology":{"ran":24,"of":30,"n_ran_checked":18,"n_instrument":6,"unverified":6,"pointer_only":1,"phrase":"24 ran (of which 2 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","official":{"repos":["dao-ailab/flash-attention"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"paper":"/paper/nearest-neighbor-zero-shot-inference","slug":"nearest-neighbor-zero-shot-inference","title":"kNN-Prompt: Nearest Neighbor Zero-Shot Inference","date":"2022-05-27","arxiv_id":"2205.13792","n_code_links":1,"syntology":null},{"paper":"/paper/what-dense-graph-do-you-need-for-self","slug":"what-dense-graph-do-you-need-for-self","title":"What Dense Graph Do You Need for Self-Attention?","date":"2022-05-27","arxiv_id":"2205.14014","n_code_links":1,"syntology":null},{"paper":null,"slug":"conditional-set-generation-using-seq2seq-1","title":"Conditional set generation using Seq2seq models","date":"2022-05-25","arxiv_id":"2205.12485","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-are-zero-shot-clinical","title":"Large Language Models are Few-Shot Clinical Information Extractors","date":"2022-05-25","arxiv_id":"2205.12689","n_code_links":0,"syntology":null},{"paper":"/paper/naturalprover-grounded-mathematical-proof","slug":"naturalprover-grounded-mathematical-proof","title":"NaturalProver: Grounded Mathematical Proof Generation with Language Models","date":"2022-05-25","arxiv_id":"2205.12910","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wellecks/naturalprover"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-understanding-label-regularization","title":"Do we need Label Regularization to Fine-tune Pre-trained Language Models?","date":"2022-05-25","arxiv_id":"2205.12428","n_code_links":0,"syntology":null},{"paper":"/paper/transcormer-transformer-for-sentence-scoring","slug":"transcormer-transformer-for-sentence-scoring","title":"Transcormer: Transformer for Sentence Scoring with Sliding Language Modeling","date":"2022-05-25","arxiv_id":"2205.12986","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/flute-figurative-language-understanding-and","slug":"flute-figurative-language-understanding-and","title":"FLUTE: Figurative Language Understanding through Textual Explanations","date":"2022-05-24","arxiv_id":"2205.12404","n_code_links":1,"syntology":null},{"paper":"/paper/garden-path-traversal-within-gpt-2","slug":"garden-path-traversal-within-gpt-2","title":"Garden-Path Traversal in GPT-2","date":"2022-05-24","arxiv_id":"2205.12302","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wjurayj/garden-path-gpt2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-human-is-human-evaluation-improving-the","title":"The Authenticity Gap in Human Evaluation","date":"2022-05-24","arxiv_id":"2205.11930","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-role-of-bidirectionality-in-language","title":"On the Role of Bidirectionality in Language Model Pre-Training","date":"2022-05-24","arxiv_id":"2205.11726","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-short-text-classification-with","title":"Improving Short Text Classification With Augmented Data Using GPT-3","date":"2022-05-23","arxiv_id":"2205.10981","n_code_links":0,"syntology":null},{"paper":"/paper/looking-for-a-handsome-carpenter-debiasing","slug":"looking-for-a-handsome-carpenter-debiasing","title":"Looking for a Handsome Carpenter! Debiasing GPT-3 Job Advertisements","date":"2022-05-23","arxiv_id":"2205.11374","n_code_links":1,"syntology":null},{"paper":null,"slug":"penguins-don-t-fly-reasoning-about-generics","title":"Penguins Don't Fly: Reasoning about Generics through Instantiations and Exceptions","date":"2022-05-23","arxiv_id":"2205.11658","n_code_links":0,"syntology":null},{"paper":null,"slug":"rl-with-kl-penalties-is-better-viewed-as","title":"RL with KL penalties is better viewed as Bayesian inference","date":"2022-05-23","arxiv_id":"2205.11275","n_code_links":0,"syntology":null},{"paper":"/paper/graphmae-self-supervised-masked-graph","slug":"graphmae-self-supervised-masked-graph","title":"GraphMAE: Self-Supervised Masked Graph Autoencoders","date":"2022-05-22","arxiv_id":"2205.10803","n_code_links":3,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thudm/graphmae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/instruction-induction-from-few-examples-to","slug":"instruction-induction-from-few-examples-to","title":"Instruction Induction: From Few Examples to Natural Language Task Descriptions","date":"2022-05-22","arxiv_id":"2205.10782","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["orhonovich/instruction-induction"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/least-to-most-prompting-enables-complex","slug":"least-to-most-prompting-enables-complex","title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","date":"2022-05-21","arxiv_id":"2205.10625","n_code_links":1,"syntology":null},{"paper":"/paper/life-after-bert-what-do-other-muppets-2","slug":"life-after-bert-what-do-other-muppets-2","title":"Life after BERT: What do Other Muppets Understand about Language?","date":"2022-05-21","arxiv_id":"2205.10696","n_code_links":1,"syntology":null},{"paper":"/paper/prototypical-calibration-for-few-shot","slug":"prototypical-calibration-for-few-shot","title":"Prototypical Calibration for Few-shot Learning of Language Models","date":"2022-05-20","arxiv_id":"2205.10183","n_code_links":1,"syntology":null},{"paper":"/paper/automated-scoring-for-reading-comprehension","slug":"automated-scoring-for-reading-comprehension","title":"Automated Scoring for Reading Comprehension via In-context BERT Tuning","date":"2022-05-19","arxiv_id":"2205.09864","n_code_links":1,"syntology":null},{"paper":"/paper/towards-understanding-gender-seniority","slug":"towards-understanding-gender-seniority","title":"Towards Understanding Gender-Seniority Compound Bias in Natural Language Generation","date":"2022-05-19","arxiv_id":"2205.09830","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluation-of-transfer-learning-for-polish-1","title":"Evaluation of Transfer Learning for Polish with a Text-to-Text Model","date":"2022-05-18","arxiv_id":"2205.08808","n_code_links":0,"syntology":null},{"paper":null,"slug":"m6-rec-generative-pretrained-language-models","title":"M6-Rec: Generative Pretrained Language Models are Open-Ended Recommender Systems","date":"2022-05-17","arxiv_id":"2205.08084","n_code_links":0,"syntology":null},{"paper":null,"slug":"heroes-villains-and-victims-and-gpt-3","title":"Heroes, Villains, and Victims, and GPT-3: Automated Extraction of Character Roles Without Training Data","date":"2022-05-16","arxiv_id":"2205.07557","n_code_links":0,"syntology":null},{"paper":"/paper/the-ai-teacher-test-measuring-the-pedagogical","slug":"the-ai-teacher-test-measuring-the-pedagogical","title":"The AI Teacher Test: Measuring the Pedagogical Ability of Blender and GPT-3 in Educational Dialogues","date":"2022-05-16","arxiv_id":"2205.07540","n_code_links":1,"syntology":null},{"paper":"/paper/what-gpt-knows-about-who-is-who-1","slug":"what-gpt-knows-about-who-is-who-1","title":"What GPT Knows About Who is Who","date":"2022-05-16","arxiv_id":"2205.07407","n_code_links":1,"syntology":null},{"paper":"/paper/naturalistic-causal-probing-for-morpho-syntax","slug":"naturalistic-causal-probing-for-morpho-syntax","title":"Naturalistic Causal Probing for Morpho-Syntax","date":"2022-05-14","arxiv_id":"2205.07043","n_code_links":1,"syntology":null},{"paper":"/paper/clinical-prompt-learning-with-frozen-language","slug":"clinical-prompt-learning-with-frozen-language","title":"Clinical Prompt Learning with Frozen Language Models","date":"2022-05-11","arxiv_id":"2205.05535","n_code_links":1,"syntology":null},{"paper":"/paper/towards-the-generation-of-musical","slug":"towards-the-generation-of-musical","title":"Towards the Generation of Musical Explanations with GPT-3","date":"2022-05-11","arxiv_id":"2206.08264","n_code_links":1,"syntology":null},{"paper":null,"slug":"object-detection-in-indian-food-platters","title":"Object Detection in Indian Food Platters using Transfer Learning with YOLOv4","date":"2022-05-10","arxiv_id":"2205.04841","n_code_links":0,"syntology":null},{"paper":null,"slug":"ratatouille-a-tool-for-novel-recipe","title":"Ratatouille: A tool for Novel Recipe Generation","date":"2022-05-10","arxiv_id":"2206.08267","n_code_links":0,"syntology":null},{"paper":"/paper/reducing-activation-recomputation-in-large","slug":"reducing-activation-recomputation-in-large","title":"Reducing Activation Recomputation in Large Transformer Models","date":"2022-05-10","arxiv_id":"2205.05198","n_code_links":4,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["NVIDIA/Megatron-LM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"paper":"/paper/unifying-language-learning-paradigms","slug":"unifying-language-learning-paradigms","title":"UL2: Unifying Language Learning Paradigms","date":"2022-05-10","arxiv_id":"2205.05131","n_code_links":2,"syntology":{"ran":15,"of":16,"n_ran_checked":15,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"multi-segment-preserving-sampling-for-deep","title":"Multi-segment preserving sampling for deep manifold sampler","date":"2022-05-09","arxiv_id":"2205.04259","n_code_links":0,"syntology":null},{"paper":"/paper/the-unreliability-of-explanations-in-few-shot","slug":"the-unreliability-of-explanations-in-few-shot","title":"The Unreliability of Explanations in Few-shot Prompting for Textual Reasoning","date":"2022-05-06","arxiv_id":"2205.03401","n_code_links":1,"syntology":null},{"paper":"/paper/when-a-sentence-does-not-introduce-a-1","slug":"when-a-sentence-does-not-introduce-a-1","title":"When a sentence does not introduce a discourse entity, Transformer-based models still sometimes refer to it","date":"2022-05-06","arxiv_id":"2205.03472","n_code_links":1,"syntology":null},{"paper":"/paper/provably-confidential-language-modelling-1","slug":"provably-confidential-language-modelling-1","title":"Provably Confidential Language Modelling","date":"2022-05-04","arxiv_id":"2205.01863","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xuandongzhao/crt"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/contrastive-learning-for-prompt-based-few","slug":"contrastive-learning-for-prompt-based-few","title":"Contrastive Learning for Prompt-Based Few-Shot Language Learners","date":"2022-05-03","arxiv_id":"2205.01308","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":4,"n_instrument":2,"unverified":5,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["yiren-jian/lm-supcon"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/mixed-effects-transformers-for-hierarchical","slug":"mixed-effects-transformers-for-hierarchical","title":"Mixed-effects transformers for hierarchical adaptation","date":"2022-05-03","arxiv_id":"2205.01749","n_code_links":1,"syntology":null},{"paper":null,"slug":"gradient-descent-stochastic-optimization-and","title":"Gradient Descent, Stochastic Optimization, and Other Tales","date":"2022-05-02","arxiv_id":"2205.00832","n_code_links":0,"syntology":null},{"paper":"/paper/opt-open-pre-trained-transformer-language","slug":"opt-open-pre-trained-transformer-language","title":"OPT: Open Pre-trained Transformer Language Models","date":"2022-05-02","arxiv_id":"2205.01068","n_code_links":11,"syntology":{"ran":14,"of":24,"n_ran_checked":14,"n_instrument":0,"unverified":10,"pointer_only":17,"phrase":"14 ran (of which 2 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","official":{"repos":["facebookresearch/metaseq"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"birds-eye-view-measuring-behavior-and-posture","title":"Birds' Eye View: Measuring Behavior and Posture of Chickens as a Metric for Their Well-Being","date":"2022-04-29","arxiv_id":"2205.00069","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-from-natural-language-feedback","title":"Training Language Models with Language Feedback","date":"2022-04-29","arxiv_id":"2204.14146","n_code_links":0,"syntology":null},{"paper":"/paper/inferring-implicit-relations-with-language","slug":"inferring-implicit-relations-with-language","title":"Inferring Implicit Relations in Complex Questions with Language Models","date":"2022-04-28","arxiv_id":"2204.13778","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-effect-of-pretraining-corpora-on-in","title":"On the Effect of Pretraining Corpora on In-context Learning by a Large-scale Language Model","date":"2022-04-28","arxiv_id":"2204.13509","n_code_links":0,"syntology":null},{"paper":null,"slug":"tailor-a-prompt-based-approach-to-attribute","title":"Tailor: A Prompt-Based Approach to Attribute-Based Controlled Text Generation","date":"2022-04-28","arxiv_id":"2204.13362","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-end-to-end-dialogue-summarization-system","title":"An End-to-End Dialogue Summarization System for Sales Calls","date":"2022-04-27","arxiv_id":"2204.12951","n_code_links":0,"syntology":null},{"paper":"/paper/emotion-aware-transformer-encoder-for-1","slug":"emotion-aware-transformer-encoder-for-1","title":"Emotion-Aware Transformer Encoder for Empathetic Dialogue Generation","date":"2022-04-24","arxiv_id":"2204.11320","n_code_links":1,"syntology":null},{"paper":"/paper/hrplanes-high-resolution-airplane-dataset-for","slug":"hrplanes-high-resolution-airplane-dataset-for","title":"A benchmark dataset for deep learning-based airplane detection: HRPlanes","date":"2022-04-22","arxiv_id":"2204.10959","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-artificial-intelligence-as-a","title":"Measuring artificial intelligence: a systematic assessment and implications for governance","date":"2022-04-21","arxiv_id":"2204.10304","n_code_links":0,"syntology":null},{"paper":"/paper/sintra-learning-an-inspiration-model-from-a","slug":"sintra-learning-an-inspiration-model-from-a","title":"SinTra: Learning an inspiration model from a single multi-track music segment","date":"2022-04-21","arxiv_id":"2204.09917","n_code_links":1,"syntology":null},{"paper":null,"slug":"codexdb-generating-code-for-processing-sql","title":"CodexDB: Generating Code for Processing SQL Queries using GPT-3 Codex","date":"2022-04-19","arxiv_id":"2204.08941","n_code_links":0,"syntology":null},{"paper":null,"slug":"impact-of-tokenization-on-language-models-an-1","title":"Impact of Tokenization on Language Models: An Analysis for Turkish","date":"2022-04-19","arxiv_id":"2204.08832","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-entity-and-tweet-characterization","title":"Zero-shot Entity and Tweet Characterization with Designed Conditional Prompts and Contexts","date":"2022-04-18","arxiv_id":"2204.08405","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-based-automatic-detection-of-2","title":"Deep Learning based Automatic Detection of Dicentric Chromosome","date":"2022-04-17","arxiv_id":"2204.08029","n_code_links":0,"syntology":null},{"paper":"/paper/auton-survival-an-open-source-package-for","slug":"auton-survival-an-open-source-package-for","title":"auton-survival: an Open-Source Package for Regression, Counterfactual Estimation, Evaluation and Phenotyping with Censored Time-to-Event Data","date":"2022-04-15","arxiv_id":"2204.07276","n_code_links":3,"syntology":null},{"paper":"/paper/mgpt-few-shot-learners-go-multilingual","slug":"mgpt-few-shot-learners-go-multilingual","title":"mGPT: Few-Shot Learners Go Multilingual","date":"2022-04-15","arxiv_id":"2204.07580","n_code_links":1,"syntology":null},{"paper":"/paper/polling-latent-opinions-a-method-for-1","slug":"polling-latent-opinions-a-method-for-1","title":"Polling Latent Opinions: A Method for Computational Sociolinguistics Using Transformer Language Models","date":"2022-04-15","arxiv_id":"2204.07483","n_code_links":1,"syntology":null},{"paper":null,"slug":"brazilian-court-documents-clustered-by","title":"Analysing similarities between legal court documents using natural language processing approaches based on Transformers","date":"2022-04-14","arxiv_id":"2204.07182","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-neox-20b-an-open-source-autoregressive-1","slug":"gpt-neox-20b-an-open-source-autoregressive-1","title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","date":"2022-04-14","arxiv_id":"2204.06745","n_code_links":11,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eleutherai/gpt-neox"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"rows-from-many-sources-enriching-row","title":"Rows from Many Sources: Enriching row completions from Wikidata with a pre-trained Language Model","date":"2022-04-14","arxiv_id":"2204.07014","n_code_links":0,"syntology":null},{"paper":"/paper/uniform-complexity-for-text-generation","slug":"uniform-complexity-for-text-generation","title":"Uniform Complexity for Text Generation","date":"2022-04-11","arxiv_id":"2204.05185","n_code_links":1,"syntology":null},{"paper":null,"slug":"foundationlayernorm-scaling-bert-and-gpt-to","title":"FoundationLayerNorm: Scaling BERT and GPT to 1,000 Layers","date":"2022-04-09","arxiv_id":"2204.04477","n_code_links":0,"syntology":null},{"paper":null,"slug":"accelerating-attention-through-gradient-based","title":"Accelerating Attention through Gradient-Based Learned Runtime Pruning","date":"2022-04-07","arxiv_id":"2204.03227","n_code_links":0,"syntology":null},{"paper":null,"slug":"bertuit-understanding-spanish-language-in","title":"BERTuit: Understanding Spanish language in Twitter through a native transformer","date":"2022-04-07","arxiv_id":"2204.03465","n_code_links":0,"syntology":null},{"paper":"/paper/testing-the-limits-of-natural-language-models","slug":"testing-the-limits-of-natural-language-models","title":"Testing the limits of natural language models for predicting human language judgments","date":"2022-04-07","arxiv_id":"2204.03592","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-infused-decoding-1","slug":"knowledge-infused-decoding-1","title":"Knowledge Infused Decoding","date":"2022-04-06","arxiv_id":"2204.03084","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/kid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/data-augmentation-for-intent-classification-1","slug":"data-augmentation-for-intent-classification-1","title":"Data Augmentation for Intent Classification with Off-the-shelf Large Language Models","date":"2022-04-05","arxiv_id":"2204.01959","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["elementai/data-augmentation-with-llms"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"effect-and-analysis-of-large-scale-language","title":"Effect and Analysis of Large-scale Language Model Rescoring on Competitive ASR Systems","date":"2022-04-01","arxiv_id":"2204.00212","n_code_links":0,"syntology":null},{"paper":"/paper/monarch-expressive-structured-matrices-for","slug":"monarch-expressive-structured-matrices-for","title":"Monarch: Expressive Structured Matrices for Efficient and Accurate Training","date":"2022-04-01","arxiv_id":"2204.00595","n_code_links":2,"syntology":{"ran":15,"of":15,"n_ran_checked":14,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hazyresearch/monarch"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"create-a-benchmark-for-chinese-short-video-1","title":"CREATE: A Benchmark for Chinese Short Video Retrieval and Title Generation","date":"2022-03-31","arxiv_id":"2203.16763","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-pre-trained-transformers-for","title":"Generative Pre-Trained Transformers for Biologically Inspired Design","date":"2022-03-31","arxiv_id":"2204.09714","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-pre-trained-language-models-for","title":"Leveraging pre-trained language models for conversational information seeking from text","date":"2022-03-31","arxiv_id":"2204.03542","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-language-models-without","slug":"transformer-language-models-without","title":"Transformer Language Models without Positional Encodings Still Learn Positional Information","date":"2022-03-30","arxiv_id":"2203.16634","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["adihaviv/nopos"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/training-compute-optimal-large-language","slug":"training-compute-optimal-large-language","title":"Training Compute-Optimal Large Language Models","date":"2022-03-29","arxiv_id":"2203.15556","n_code_links":2,"syntology":{"ran":8,"of":11,"n_ran_checked":5,"n_instrument":3,"unverified":3,"pointer_only":4,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/efficient-vdvae-less-is-more","slug":"efficient-vdvae-less-is-more","title":"Efficient-VDVAE: Less is more","date":"2022-03-25","arxiv_id":"2203.13751","n_code_links":1,"syntology":null},{"paper":"/paper/bailando-3d-dance-generation-by-actor-critic","slug":"bailando-3d-dance-generation-by-actor-critic","title":"Bailando: 3D Dance Generation by Actor-Critic GPT with Choreographic Memory","date":"2022-03-24","arxiv_id":"2203.13055","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lisiyao21/bailando"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ernie-sparse-learning-hierarchical-efficient-1","title":"ERNIE-SPARSE: Learning Hierarchical Efficient Transformer Through Regularized Self-Attention","date":"2022-03-23","arxiv_id":"2203.12276","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervision-through-random-segments-with","title":"Self-supervision through Random Segments with Autoregressive Coding (RandSAC)","date":"2022-03-22","arxiv_id":"2203.12054","n_code_links":0,"syntology":null},{"paper":"/paper/a-slot-is-not-built-in-one-utterance-spoken-1","slug":"a-slot-is-not-built-in-one-utterance-spoken-1","title":"A Slot Is Not Built in One Utterance: Spoken Language Dialogs with Sub-Slots","date":"2022-03-21","arxiv_id":"2203.10759","n_code_links":1,"syntology":null},{"paper":null,"slug":"compression-of-generative-pre-trained","title":"Compression of Generative Pre-trained Language Models via Quantization","date":"2022-03-21","arxiv_id":"2203.10705","n_code_links":0,"syntology":null},{"paper":"/paper/dependency-based-mixture-language-models","slug":"dependency-based-mixture-language-models","title":"Dependency-based Mixture Language Models","date":"2022-03-19","arxiv_id":"2203.10256","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["fadedcosine/dependency-guided-neural-text-generation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"analysis-and-adaptation-of-yolov4-for-object","title":"Analysis and Adaptation of YOLOv4 for Object Detection in Aerial Images","date":"2022-03-18","arxiv_id":"2203.10194","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-you-robert-or-roberta-deceiving-online","title":"Are You Robert or RoBERTa? Deceiving Online Authorship Attribution Models Using Neural Text Generators","date":"2022-03-18","arxiv_id":"2203.09813","n_code_links":0,"syntology":null},{"paper":"/paper/thinking-about-gpt-3-in-context-learning-for","slug":"thinking-about-gpt-3-in-context-learning-for","title":"Thinking about GPT-3 In-Context Learning for Biomedical IE? Think Again","date":"2022-03-16","arxiv_id":"2203.08410","n_code_links":1,"syntology":null},{"paper":"/paper/do-language-models-plagiarize","slug":"do-language-models-plagiarize","title":"Do Language Models Plagiarize?","date":"2022-03-15","arxiv_id":"2203.07618","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-ghost-in-the-machine-has-an-american","title":"The Ghost in the Machine has an American accent: value conflict in GPT-3","date":"2022-03-15","arxiv_id":"2203.07785","n_code_links":0,"syntology":null},{"paper":null,"slug":"contrastive-visual-semantic-pretraining","title":"Contrastive Visual Semantic Pretraining Magnifies the Semantics of Natural Language Representations","date":"2022-03-14","arxiv_id":"2203.07511","n_code_links":0,"syntology":null},{"paper":"/paper/grips-gradient-free-edit-based-instruction","slug":"grips-gradient-free-edit-based-instruction","title":"GrIPS: Gradient-free, Edit-based Instruction Search for Prompting Large Language Models","date":"2022-03-14","arxiv_id":"2203.07281","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["archiki/grips"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/vast-the-valence-assessing-semantics-test-for","slug":"vast-the-valence-assessing-semantics-test-for","title":"VAST: The Valence-Assessing Semantics Test for Contextualizing Language Models","date":"2022-03-14","arxiv_id":"2203.07504","n_code_links":1,"syntology":null},{"paper":"/paper/elle-efficient-lifelong-pre-training-for-1","slug":"elle-efficient-lifelong-pre-training-for-1","title":"ELLE: Efficient Lifelong Pre-training for Emerging Data","date":"2022-03-12","arxiv_id":"2203.06311","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":1,"n_instrument":3,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thunlp/elle"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/block-sparse-adversarial-attack-to-fool","slug":"block-sparse-adversarial-attack-to-fool","title":"Block-Sparse Adversarial Attack to Fool Transformer-Based Text Classifiers","date":"2022-03-11","arxiv_id":"2203.05948","n_code_links":1,"syntology":null}],"record_sha256":"c7fda612f80a44d074262e41261cb83daffb8f7b32be930850c9ec808822241a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}