{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/inverse-square-root-schedule/papers/5","list_of":"/method/inverse-square-root-schedule","method":"Inverse Square Root Schedule","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":8,"rows_per_page":100,"rows":[401,500],"of":702,"counts":{"archive_papers_tagged":702,"with_a_code_link":349,"where_syntology_ran_a_sample":97,"not_listed_spam_title":0,"listed":702,"listed_where_code_ran":97,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":83,"every_run_a_failure_of_syntologys_instrument":14,"listed_with_a_run_with_no_instrument_failure":83,"listed_every_run_a_failure_of_syntologys_instrument":14,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/inverse-square-root-schedule","prev":"/method/inverse-square-root-schedule/papers/4","next":"/method/inverse-square-root-schedule/papers/6","papers":[{"paper":"/paper/uncontrolled-lexical-exposure-leads-to","slug":"uncontrolled-lexical-exposure-leads-to","title":"Uncontrolled Lexical Exposure Leads to Overestimation of Compositional Generalization in Pretrained Models","date":"2022-12-21","arxiv_id":"2212.10769","n_code_links":1,"syntology":null},{"paper":"/paper/bygpt5-end-to-end-style-conditioned-poetry","slug":"bygpt5-end-to-end-style-conditioned-poetry","title":"ByGPT5: End-to-End Style-conditioned Poetry Generation with Token-free Language Models","date":"2022-12-20","arxiv_id":"2212.10474","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["potamides/uniformers"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/do-language-models-have-coherent-mental","slug":"do-language-models-have-coherent-mental","title":"Do language models have coherent mental models of everyday things?","date":"2022-12-20","arxiv_id":"2212.10029","n_code_links":1,"syntology":{"ran":0,"of":5,"n_ran_checked":0,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"0 ran · 5 unverified","official":{"repos":["allenai/everyday-things"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"paper":null,"slug":"krona-parameter-efficient-tuning-with","title":"KronA: Parameter Efficient Tuning with Kronecker Adapter","date":"2022-12-20","arxiv_id":"2212.10650","n_code_links":0,"syntology":null},{"paper":null,"slug":"receptive-field-alignment-enables-transformer","title":"Dissecting Transformer Length Extrapolation via the Lens of Receptive Field Analysis","date":"2022-12-20","arxiv_id":"2212.10356","n_code_links":0,"syntology":null},{"paper":"/paper/t-projection-high-quality-annotation","slug":"t-projection-high-quality-annotation","title":"T-Projection: High Quality Annotation Projection for Sequence Labeling Tasks","date":"2022-12-20","arxiv_id":"2212.10548","n_code_links":2,"syntology":null},{"paper":"/paper/do-conll-2003-named-entity-taggers-still-work","slug":"do-conll-2003-named-entity-taggers-still-work","title":"Do CoNLL-2003 Named Entity Taggers Still Work Well in 2023?","date":"2022-12-19","arxiv_id":"2212.09747","n_code_links":1,"syntology":null},{"paper":null,"slug":"miga-a-unified-multi-task-generation","title":"MIGA: A Unified Multi-task Generation Framework for Conversational Text-to-SQL","date":"2022-12-19","arxiv_id":"2212.09278","n_code_links":0,"syntology":null},{"paper":null,"slug":"mu-2-slam-multitask-multilingual-speech-and","title":"Mu$^{2}$SLAM: Multitask, Multilingual Speech and Language Models","date":"2022-12-19","arxiv_id":"2212.09553","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-sequence-to-sequence-models-for","title":"Multilingual Sequence-to-Sequence Models for Hebrew NLP","date":"2022-12-19","arxiv_id":"2212.09682","n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-small-language-models-to-reason","title":"Teaching Small Language Models to Reason","date":"2022-12-16","arxiv_id":"2212.08410","n_code_links":0,"syntology":null},{"paper":"/paper/fido-fusion-in-decoder-optimized-for-stronger","slug":"fido-fusion-in-decoder-optimized-for-stronger","title":"FiDO: Fusion-in-Decoder optimized for stronger performance and faster inference","date":"2022-12-15","arxiv_id":"2212.08153","n_code_links":0,"syntology":null},{"paper":"/paper/visually-augmented-pretrained-language-models","slug":"visually-augmented-pretrained-language-models","title":"Visually-augmented pretrained language models for NLP tasks without images","date":"2022-12-15","arxiv_id":"2212.07937","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-byte-and-wordpiece-level-models","title":"Evaluating Byte and Wordpiece Level Models for Massively Multilingual Semantic Parsing","date":"2022-12-14","arxiv_id":"2212.07223","n_code_links":0,"syntology":null},{"paper":"/paper/t5score-discriminative-fine-tuning-of","slug":"t5score-discriminative-fine-tuning-of","title":"T5Score: Discriminative Fine-tuning of Generative Evaluation Metrics","date":"2022-12-12","arxiv_id":"2212.05726","n_code_links":2,"syntology":null},{"paper":"/paper/sparse-upcycling-training-mixture-of-experts","slug":"sparse-upcycling-training-mixture-of-experts","title":"Sparse Upcycling: Training Mixture-of-Experts from Dense Checkpoints","date":"2022-12-09","arxiv_id":"2212.05055","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/vmoe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"trbllmaker-transformer-reads-between-lyrics","title":"TRBLLmaker -- Transformer Reads Between Lyrics Lines maker","date":"2022-12-09","arxiv_id":"2212.04917","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-multimodal-transformers-for","slug":"hierarchical-multimodal-transformers-for","title":"Hierarchical multimodal transformers for Multi-Page DocVQA","date":"2022-12-07","arxiv_id":"2212.05935","n_code_links":1,"syntology":null},{"paper":null,"slug":"controlled-text-generation-using-t5-based","title":"Controlled Text Generation using T5 based Encoder-Decoder Soft Prompt Tuning and Analysis of the Utility of Generated Text in AI","date":"2022-12-06","arxiv_id":"2212.02924","n_code_links":0,"syntology":null},{"paper":null,"slug":"languages-you-know-influence-those-you-learn","title":"Languages You Know Influence Those You Learn: Impact of Language Characteristics on Multi-Lingual Text-to-Text Transfer","date":"2022-12-04","arxiv_id":"2212.01757","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-memory-transformer-for-processing-long","title":"Global memory transformer for processing long documents","date":"2022-12-03","arxiv_id":"2212.01650","n_code_links":0,"syntology":null},{"paper":"/paper/data-efficient-finetuning-using-cross-task","slug":"data-efficient-finetuning-using-cross-task","title":"Data-Efficient Finetuning Using Cross-Task Nearest Neighbors","date":"2022-12-01","arxiv_id":"2212.00196","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allenai/data-efficient-finetuning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"detect-localize-repair-a-unified-framework","title":"Detect-Localize-Repair: A Unified Framework for Learning to Debug with CodeT5","date":"2022-11-27","arxiv_id":"2211.14875","n_code_links":0,"syntology":null},{"paper":"/paper/coreference-resolution-through-a-seq2seq","slug":"coreference-resolution-through-a-seq2seq","title":"Coreference Resolution through a seq2seq Transition-Based System","date":"2022-11-22","arxiv_id":"2211.12142","n_code_links":1,"syntology":null},{"paper":null,"slug":"hypertuning-toward-adapting-large-language","title":"HyperTuning: Toward Adapting Large Language Models without Back-propagation","date":"2022-11-22","arxiv_id":"2211.12485","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-self-consistency-and-performance-of","title":"Enhancing Self-Consistency and Performance of Pre-Trained Language Models through Natural Language Inference","date":"2022-11-21","arxiv_id":"2211.11875","n_code_links":0,"syntology":null},{"paper":"/paper/unifiedabsa-a-unified-absa-framework-based-on","slug":"unifiedabsa-a-unified-absa-framework-based-on","title":"UnifiedABSA: A Unified ABSA Framework Based on Multi-task Instruction Tuning","date":"2022-11-20","arxiv_id":"2211.10986","n_code_links":0,"syntology":null},{"paper":"/paper/glami-1m-a-multilingual-image-text-fashion-1","slug":"glami-1m-a-multilingual-image-text-fashion-1","title":"GLAMI-1M: A Multilingual Image-Text Fashion Dataset","date":"2022-11-17","arxiv_id":"2211.14451","n_code_links":1,"syntology":null},{"paper":"/paper/unified-question-answering-in-slovene","slug":"unified-question-answering-in-slovene","title":"Unified Question Answering in Slovene","date":"2022-11-16","arxiv_id":"2211.09159","n_code_links":1,"syntology":null},{"paper":"/paper/breakpoint-transformers-for-modeling-and","slug":"breakpoint-transformers-for-modeling-and","title":"Breakpoint Transformers for Modeling and Tracking Intermediate Beliefs","date":"2022-11-15","arxiv_id":"2211.07950","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":2,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["allenai/situation_modeling"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"empowering-language-models-with-knowledge","title":"Empowering Language Models with Knowledge Graph Reasoning for Question Answering","date":"2022-11-15","arxiv_id":"2211.08380","n_code_links":0,"syntology":null},{"paper":"/paper/cst5-data-augmentation-for-code-switched-1","slug":"cst5-data-augmentation-for-code-switched-1","title":"CST5: Data Augmentation for Code-Switched Semantic Parsing","date":"2022-11-14","arxiv_id":"2211.07514","n_code_links":1,"syntology":null},{"paper":"/paper/technological-taxonomies-for-hypernym-and","slug":"technological-taxonomies-for-hypernym-and","title":"Technological taxonomies for hypernym and hyponym retrieval in patent texts","date":"2022-11-14","arxiv_id":"2212.06039","n_code_links":1,"syntology":null},{"paper":null,"slug":"docut5-seq2seq-sql-generation-with-table","title":"DocuT5: Seq2seq SQL Generation with Table Documentation","date":"2022-11-11","arxiv_id":"2211.06193","n_code_links":0,"syntology":null},{"paper":null,"slug":"assistive-completion-of-agrammatic-aphasic","title":"Assistive Completion of Agrammatic Aphasic Sentences: A Transfer Learning Approach using Neurolinguistics-based Synthetic Dataset","date":"2022-11-10","arxiv_id":"2211.05557","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-with-controllable","title":"Large Language Models with Controllable Working Memory","date":"2022-11-09","arxiv_id":"2211.05110","n_code_links":0,"syntology":null},{"paper":"/paper/conciseness-an-overlooked-language-task","slug":"conciseness-an-overlooked-language-task","title":"Conciseness: An Overlooked Language Task","date":"2022-11-08","arxiv_id":"2211.04126","n_code_links":0,"syntology":null},{"paper":"/paper/crosslingual-generalization-through-multitask","slug":"crosslingual-generalization-through-multitask","title":"Crosslingual Generalization through Multitask Finetuning","date":"2022-11-03","arxiv_id":"2211.01786","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bigscience-workshop/xmtf"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"using-large-pre-trained-language-model-to","title":"Using Large Pre-Trained Language Model to Assist FDA in Premarket Medical Device","date":"2022-11-03","arxiv_id":"2212.01217","n_code_links":0,"syntology":null},{"paper":"/paper/ediffi-text-to-image-diffusion-models-with-an","slug":"ediffi-text-to-image-diffusion-models-with-an","title":"eDiff-I: Text-to-Image Diffusion Models with an Ensemble of Expert Denoisers","date":"2022-11-02","arxiv_id":"2211.01324","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":10,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"frsum-towards-faithful-abstractive-1","title":"FRSUM: Towards Faithful Abstractive Summarization via Enhancing Factual Robustness","date":"2022-11-01","arxiv_id":"2211.00294","n_code_links":0,"syntology":null},{"paper":null,"slug":"preserving-in-context-learning-ability-in","title":"Two-stage LLM Fine-tuning with Less Specialization and More Generalization","date":"2022-11-01","arxiv_id":"2211.00635","n_code_links":0,"syntology":null},{"paper":"/paper/t5lephone-bridging-speech-and-text-self","slug":"t5lephone-bridging-speech-and-text-self","title":"T5lephone: Bridging Speech and Text Self-supervised Models for Spoken Language Understanding via Phoneme level T5","date":"2022-11-01","arxiv_id":"2211.00586","n_code_links":1,"syntology":null},{"paper":null,"slug":"beyond-english-centric-bitexts-for-better","title":"Beyond English-Centric Bitexts for Better Multilingual Language Representation Learning","date":"2022-10-26","arxiv_id":"2210.14867","n_code_links":0,"syntology":null},{"paper":"/paper/leveraging-affirmative-interpretations-from","slug":"leveraging-affirmative-interpretations-from","title":"Leveraging Affirmative Interpretations from Negation Improves Natural Language Understanding","date":"2022-10-26","arxiv_id":"2210.14486","n_code_links":1,"syntology":null},{"paper":"/paper/amos-an-adam-style-optimizer-with-adaptive","slug":"amos-an-adam-style-optimizer-with-adaptive","title":"Amos: An Adam-style Optimizer with Adaptive Weight Decay towards Model-Oriented Scale","date":"2022-10-21","arxiv_id":"2210.11693","n_code_links":1,"syntology":{"ran":14,"of":18,"n_ran_checked":14,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["google-research/jestimator"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/decoding-a-neural-retriever-s-latent-space","slug":"decoding-a-neural-retriever-s-latent-space","title":"Decoding a Neural Retriever's Latent Space for Query Suggestion","date":"2022-10-21","arxiv_id":"2210.12084","n_code_links":1,"syntology":null},{"paper":"/paper/sling-sino-linguistic-evaluation-of-large","slug":"sling-sino-linguistic-evaluation-of-large","title":"SLING: Sino Linguistic Evaluation of Large Language Models","date":"2022-10-21","arxiv_id":"2210.11689","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yixiao-song/sling_data_code"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-instruction-finetuned-language-models","slug":"scaling-instruction-finetuned-language-models","title":"Scaling Instruction-Finetuned Language Models","date":"2022-10-20","arxiv_id":"2210.11416","n_code_links":9,"syntology":{"ran":8,"of":17,"n_ran_checked":1,"n_instrument":7,"unverified":9,"pointer_only":2,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 9 unverified","official":null}},{"paper":"/paper/self-supervised-graph-masking-pre-training","slug":"self-supervised-graph-masking-pre-training","title":"Self-supervised Graph Masking Pre-training for Graph-to-Text Generation","date":"2022-10-19","arxiv_id":"2210.10599","n_code_links":1,"syntology":null},{"paper":"/paper/swinv2-imagen-hierarchical-vision-transformer","slug":"swinv2-imagen-hierarchical-vision-transformer","title":"Swinv2-Imagen: Hierarchical Vision Transformer Diffusion Models for Text-to-Image Generation","date":"2022-10-18","arxiv_id":"2210.09549","n_code_links":0,"syntology":null},{"paper":"/paper/john-is-50-years-old-can-his-son-be-65","slug":"john-is-50-years-old-can-his-son-be-65","title":"\"John is 50 years old, can his son be 65?\" Evaluating NLP Models' Understanding of Feasibility","date":"2022-10-14","arxiv_id":"2210.07471","n_code_links":1,"syntology":null},{"paper":null,"slug":"self-repetition-in-abstractive-neural","title":"Self-Repetition in Abstractive Neural Summarizers","date":"2022-10-14","arxiv_id":"2210.08145","n_code_links":0,"syntology":null},{"paper":null,"slug":"tone-prediction-and-orthographic-conversion","title":"Tone prediction and orthographic conversion for Basaa","date":"2022-10-13","arxiv_id":"2210.06986","n_code_links":0,"syntology":null},{"paper":"/paper/frustratingly-simple-entity-tracking-with","slug":"frustratingly-simple-entity-tracking-with","title":"Entity Tracking via Effective Use of Multi-Task Learning Model and Mention-guided Decoding","date":"2022-10-12","arxiv_id":"2210.06444","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["iamjanvijay/meet","iamjanvijay/set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/instruction-tuning-for-few-shot-aspect-based","slug":"instruction-tuning-for-few-shot-aspect-based","title":"Instruction Tuning for Few-Shot Aspect-Based Sentiment Analysis","date":"2022-10-12","arxiv_id":"2210.06629","n_code_links":1,"syntology":null},{"paper":null,"slug":"rankt5-fine-tuning-t5-for-text-ranking-with","title":"RankT5: Fine-Tuning T5 for Text Ranking with Ranking Losses","date":"2022-10-12","arxiv_id":"2210.10634","n_code_links":0,"syntology":null},{"paper":"/paper/are-pretrained-multilingual-models-equally-1","slug":"are-pretrained-multilingual-models-equally-1","title":"Are Pretrained Multilingual Models Equally Fair Across Languages?","date":"2022-10-11","arxiv_id":"2210.05457","n_code_links":1,"syntology":null},{"paper":null,"slug":"reflection-of-thought-inversely-eliciting","title":"Reflection of Thought: Inversely Eliciting Numerical Reasoning in Language Models via Solving Linear Systems","date":"2022-10-11","arxiv_id":"2210.05075","n_code_links":0,"syntology":null},{"paper":"/paper/t5-for-hate-speech-augmented-data-and","slug":"t5-for-hate-speech-augmented-data-and","title":"T5 for Hate Speech, Augmented Data and Ensemble","date":"2022-10-11","arxiv_id":"2210.05480","n_code_links":1,"syntology":null},{"paper":"/paper/asdot-any-shot-data-to-text-generation-with","slug":"asdot-any-shot-data-to-text-generation-with","title":"ASDOT: Any-Shot Data-to-Text Generation with Pretrained Language Models","date":"2022-10-09","arxiv_id":"2210.04325","n_code_links":1,"syntology":null},{"paper":"/paper/chard-clinical-health-aware-reasoning-across","slug":"chard-clinical-health-aware-reasoning-across","title":"CHARD: Clinical Health-Aware Reasoning Across Dimensions for Text Generation Models","date":"2022-10-09","arxiv_id":"2210.04191","n_code_links":1,"syntology":null},{"paper":null,"slug":"improve-transformer-pre-training-with","title":"Better Pre-Training by Reducing Representation Confusion","date":"2022-10-09","arxiv_id":"2210.04246","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-quizzes-to-support-training-on","title":"Generating Quizzes to Support Training on Quality Management and Assurance in Space Science and Engineering","date":"2022-10-07","arxiv_id":"2210.03427","n_code_links":0,"syntology":null},{"paper":"/paper/how-large-language-models-are-transforming","slug":"how-large-language-models-are-transforming","title":"How Large Language Models are Transforming Machine-Paraphrased Plagiarism","date":"2022-10-07","arxiv_id":"2210.03568","n_code_links":3,"syntology":null},{"paper":"/paper/nmtsloth-understanding-and-testing-efficiency","slug":"nmtsloth-understanding-and-testing-efficiency","title":"LLMEffiChecker: Understanding and Testing Efficiency Degradation of Large Language Models","date":"2022-10-07","arxiv_id":"2210.03696","n_code_links":1,"syntology":null},{"paper":"/paper/promptkg-a-prompt-learning-framework-for","slug":"promptkg-a-prompt-learning-framework-for","title":"LambdaKG: A Library for Pre-trained Language Model-Based Knowledge Graph Embeddings","date":"2022-10-01","arxiv_id":"2210.00305","n_code_links":2,"syntology":null},{"paper":null,"slug":"bidirectional-language-models-are-also-few","title":"Bidirectional Language Models Are Also Few-shot Learners","date":"2022-09-29","arxiv_id":"2209.14500","n_code_links":0,"syntology":null},{"paper":"/paper/wikides-a-wikipedia-based-dataset-for","slug":"wikides-a-wikipedia-based-dataset-for","title":"WikiDes: A Wikipedia-Based Dataset for Generating Short Descriptions from Paragraphs","date":"2022-09-27","arxiv_id":"2209.13101","n_code_links":1,"syntology":null},{"paper":"/paper/application-of-deep-learning-in-generating-1","slug":"application-of-deep-learning-in-generating-1","title":"Application of Deep Learning in Generating Structured Radiology Reports: A Transformer-Based Technique","date":"2022-09-25","arxiv_id":"2209.12177","n_code_links":1,"syntology":null},{"paper":"/paper/et5-a-novel-end-to-end-framework-for","slug":"et5-a-novel-end-to-end-framework-for","title":"ET5: A Novel End-to-end Framework for Conversational Machine Reading Comprehension","date":"2022-09-23","arxiv_id":"2209.11484","n_code_links":1,"syntology":null},{"paper":"/paper/xf2t-cross-lingual-fact-to-text-generation","slug":"xf2t-cross-lingual-fact-to-text-generation","title":"XF2T: Cross-lingual Fact-to-Text Generation for Low-Resource Languages","date":"2022-09-22","arxiv_id":"2209.11252","n_code_links":0,"syntology":null},{"paper":null,"slug":"t5ql-taming-language-models-for-sql","title":"T5QL: Taming language models for SQL generation","date":"2022-09-21","arxiv_id":"2209.10254","n_code_links":0,"syntology":null},{"paper":null,"slug":"chain-of-explanation-new-prompting-method-to","title":"Chain of Explanation: New Prompting Method to Generate Higher Quality Natural Language Explanation for Implicit Hate Speech","date":"2022-09-11","arxiv_id":"2209.04889","n_code_links":0,"syntology":null},{"paper":null,"slug":"simple-and-effective-gradient-based-tuning-of","title":"Simple and Effective Gradient-Based Tuning of Sequence-to-Sequence Models","date":"2022-09-10","arxiv_id":"2209.04683","n_code_links":0,"syntology":null},{"paper":"/paper/idiapers-causal-news-corpus-2022-extracting","slug":"idiapers-causal-news-corpus-2022-extracting","title":"IDIAPers @ Causal News Corpus 2022: Extracting Cause-Effect-Signal Triplets via Pre-trained Autoregressive Language Model","date":"2022-09-08","arxiv_id":"2209.03891","n_code_links":1,"syntology":null},{"paper":"/paper/mdia-a-benchmark-for-multilingual-dialogue","slug":"mdia-a-benchmark-for-multilingual-dialogue","title":"MDIA: A Benchmark for Multilingual Dialogue Generation in 46 Languages","date":"2022-08-27","arxiv_id":"2208.13078","n_code_links":1,"syntology":null},{"paper":"/paper/autoqgs-auto-prompt-for-low-resource","slug":"autoqgs-auto-prompt-for-low-resource","title":"AutoQGS: Auto-Prompt for Low-Resource Knowledge-based Question Generation from SPARQL","date":"2022-08-26","arxiv_id":"2208.12461","n_code_links":1,"syntology":null},{"paper":null,"slug":"training-a-t5-using-lab-sized-resources","title":"Training a T5 Using Lab-sized Resources","date":"2022-08-25","arxiv_id":"2208.12097","n_code_links":0,"syntology":null},{"paper":"/paper/diverse-title-generation-for-stack-overflow","slug":"diverse-title-generation-for-stack-overflow","title":"Diverse Title Generation for Stack Overflow Posts with Multiple Sampling Enhanced Transformer","date":"2022-08-24","arxiv_id":"2208.11523","n_code_links":1,"syntology":null},{"paper":"/paper/mulzdg-multilingual-code-switching-framework","slug":"mulzdg-multilingual-code-switching-framework","title":"MulZDG: Multilingual Code-Switching Framework for Zero-shot Dialogue Generation","date":"2022-08-18","arxiv_id":"2208.08629","n_code_links":1,"syntology":null},{"paper":null,"slug":"summarizing-patients-problems-from-hospital","title":"Summarizing Patients Problems from Hospital Progress Notes Using Pre-trained Sequence-to-Sequence Models","date":"2022-08-17","arxiv_id":"2208.08408","n_code_links":0,"syntology":null},{"paper":null,"slug":"continuous-active-learning-using-pretrained","title":"Continuous Active Learning Using Pretrained Transformers","date":"2022-08-15","arxiv_id":"2208.06955","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-boring-yet-effective-approach-for-the","title":"A Boring-yet-effective Approach for the Product Ranking Task of the Amazon KDD Cup 2022","date":"2022-08-09","arxiv_id":"2208.06264","n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-knowledge-bank-for-pretrained","title":"Neural Knowledge Bank for Pretrained Transformers","date":"2022-07-31","arxiv_id":"2208.00399","n_code_links":0,"syntology":null},{"paper":"/paper/sequence-to-sequence-pretraining-for-a-less","slug":"sequence-to-sequence-pretraining-for-a-less","title":"Sequence to sequence pretraining for a less-resourced Slovenian language","date":"2022-07-28","arxiv_id":"2207.13988","n_code_links":1,"syntology":null},{"paper":"/paper/no-more-fine-tuning-an-experimental","slug":"no-more-fine-tuning-an-experimental","title":"No More Fine-Tuning? An Experimental Evaluation of Prompt Tuning in Code Intelligence","date":"2022-07-24","arxiv_id":"2207.11680","n_code_links":1,"syntology":{"ran":7,"of":14,"n_ran_checked":6,"n_instrument":1,"unverified":7,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["adf1178/pt4code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"context-based-lemmatizer-for-polish-language","title":"Context based lemmatizer for Polish language","date":"2022-07-23","arxiv_id":"2207.11565","n_code_links":0,"syntology":null},{"paper":null,"slug":"effectiveness-of-french-language-models-on","title":"Effectiveness of French Language Models on Abstractive Dialogue Summarization Task","date":"2022-07-17","arxiv_id":"2207.08305","n_code_links":0,"syntology":null},{"paper":"/paper/doccoder-generating-code-by-retrieving-and","slug":"doccoder-generating-code-by-retrieving-and","title":"DocPrompting: Generating Code by Retrieving the Docs","date":"2022-07-13","arxiv_id":"2207.05987","n_code_links":2,"syntology":{"ran":4,"of":9,"n_ran_checked":2,"n_instrument":2,"unverified":5,"pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["shuyanzhou/doccoder","shuyanzhou/docprompting"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/re2g-retrieve-rerank-generate-2","slug":"re2g-retrieve-rerank-generate-2","title":"Re2G: Retrieve, Rerank, Generate","date":"2022-07-13","arxiv_id":"2207.06300","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ibm/kgi-slot-filling"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-diversify-for-product-question","title":"Learning to Diversify for Product Question Generation","date":"2022-07-06","arxiv_id":"2207.02534","n_code_links":0,"syntology":null},{"paper":"/paper/coderl-mastering-code-generation-through","slug":"coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.01780","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["salesforce/coderl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/generating-repetitions-with-appropriate","slug":"generating-repetitions-with-appropriate","title":"Generating Repetitions with Appropriate Repeated Words","date":"2022-07-03","arxiv_id":"2207.00929","n_code_links":1,"syntology":null},{"paper":"/paper/is-neural-language-acquisition-similar-to","slug":"is-neural-language-acquisition-similar-to","title":"Is neural language acquisition similar to natural? A chronological probing study","date":"2022-07-01","arxiv_id":"2207.00560","n_code_links":1,"syntology":null},{"paper":"/paper/argumentative-text-generation-in-economic","slug":"argumentative-text-generation-in-economic","title":"Argumentative Text Generation in Economic Domain","date":"2022-06-18","arxiv_id":"2206.09251","n_code_links":1,"syntology":null},{"paper":null,"slug":"automatic-summarization-of-russian-texts","title":"Automatic Summarization of Russian Texts: Comparison of Extractive and Abstractive Methods","date":"2022-06-18","arxiv_id":"2206.09253","n_code_links":0,"syntology":null},{"paper":null,"slug":"alexa-teacher-model-pretraining-and","title":"Alexa Teacher Model: Pretraining and Distilling Multi-Billion-Parameter Encoders for Natural Language Understanding Systems","date":"2022-06-15","arxiv_id":"2206.07808","n_code_links":0,"syntology":null},{"paper":"/paper/natgen-generative-pre-training-by","slug":"natgen-generative-pre-training-by","title":"NatGen: Generative pre-training by \"Naturalizing\" source code","date":"2022-06-15","arxiv_id":"2206.07585","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["saikat107/natgen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lst-ladder-side-tuning-for-parameter-and","slug":"lst-ladder-side-tuning-for-parameter-and","title":"LST: Ladder Side-Tuning for Parameter and Memory Efficient Transfer Learning","date":"2022-06-13","arxiv_id":"2206.06522","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ylsung/ladder-side-tuning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"e0d6679c79899e0878648f145d1865b2e2d6dc96a20daad5c274316c84972fb6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}