{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/weight-decay/papers/61","list_of":"/method/weight-decay","method":"Weight Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":61,"pages_in_order":108,"rows_per_page":100,"rows":[6001,6100],"of":10713,"counts":{"archive_papers_tagged":10713,"with_a_code_link":4533,"where_syntology_ran_a_sample":1291,"not_listed_spam_title":0,"listed":10713,"listed_where_code_ran":1291,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1064,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1064,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/weight-decay","prev":"/method/weight-decay/papers/60","next":"/method/weight-decay/papers/62","papers":[{"paper":"/paper/foundation-transformers","slug":"foundation-transformers","title":"Foundation Transformers","date":"2022-10-12","arxiv_id":"2210.06423","n_code_links":4,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"gmp-well-tuned-global-magnitude-pruning-can","title":"GMP*: Well-Tuned Gradual Magnitude Pruning Can Outperform Most BERT-Pruning Methods","date":"2022-10-12","arxiv_id":"2210.06384","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-text-style-transfer-via-style-masked","title":"On Text Style Transfer via Style Masked Language Models","date":"2022-10-12","arxiv_id":"2210.06394","n_code_links":0,"syntology":null},{"paper":"/paper/predictive-querying-for-autoregressive-neural","slug":"predictive-querying-for-autoregressive-neural","title":"Predictive Querying for Autoregressive Neural Sequence Models","date":"2022-10-12","arxiv_id":"2210.06464","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":2,"n_instrument":5,"unverified":6,"pointer_only":0,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["ajboyd2/prob_seq_queries"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/probing-commonsense-knowledge-in-pre-trained","slug":"probing-commonsense-knowledge-in-pre-trained","title":"Probing Commonsense Knowledge in Pre-trained Language Models with Sense-level Precision and Expanded Vocabulary","date":"2022-10-12","arxiv_id":"2210.06376","n_code_links":1,"syntology":null},{"paper":null,"slug":"rankt5-fine-tuning-t5-for-text-ranking-with","title":"RankT5: Fine-Tuning T5 for Text Ranking with Ranking Losses","date":"2022-10-12","arxiv_id":"2210.10634","n_code_links":0,"syntology":null},{"paper":null,"slug":"sumbot-summarizing-context-in-open-domain","title":"SUMBot: Summarizing Context in Open-Domain Dialogue Systems","date":"2022-10-12","arxiv_id":"2210.06496","n_code_links":0,"syntology":null},{"paper":"/paper/a-win-win-deal-towards-sparse-and-robust-pre","slug":"a-win-win-deal-towards-sparse-and-robust-pre","title":"A Win-win Deal: Towards Sparse and Robust Pre-trained Language Models","date":"2022-10-11","arxiv_id":"2210.05211","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["llyx97/sparse-and-robust-plm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-exploration-of-hierarchical-attention","title":"An Exploration of Hierarchical Attention Transformers for Efficient Long Document Classification","date":"2022-10-11","arxiv_id":"2210.05529","n_code_links":0,"syntology":null},{"paper":null,"slug":"clip-also-understands-text-prompting-clip-for","title":"CLIP also Understands Text: Prompting CLIP for Phrase Understanding","date":"2022-10-11","arxiv_id":"2210.05836","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-bert-has-an-accent-evaluating","title":"Multilingual BERT has an accent: Evaluating English influences on fluency in multilingual models","date":"2022-10-11","arxiv_id":"2210.05619","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-interpolation-of-contextualized-term","slug":"on-the-interpolation-of-contextualized-term","title":"On the Interpolation of Contextualized Term-based Ranking with BM25 for Query-by-Example Retrieval","date":"2022-10-11","arxiv_id":"2210.05512","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-use-of-semantically-aligned-speech","title":"On the Use of Semantically-Aligned Speech Representations for Spoken Language Understanding","date":"2022-10-11","arxiv_id":"2210.05291","n_code_links":0,"syntology":null},{"paper":"/paper/vote-n-rank-revision-of-benchmarking-with","slug":"vote-n-rank-revision-of-benchmarking-with","title":"Vote'n'Rank: Revision of Benchmarking with Social Choice Theory","date":"2022-10-11","arxiv_id":"2210.05769","n_code_links":1,"syntology":{"ran":6,"of":17,"n_ran_checked":6,"n_instrument":0,"unverified":11,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["pragmaticslab/vote_and_rank"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":11,"ran_from_kinds":["official"]}}},{"paper":"/paper/deptweet-a-typology-for-social-media-texts-to","slug":"deptweet-a-typology-for-social-media-texts-to","title":"DEPTWEET: A Typology for Social Media Texts to Detect Depression Severities","date":"2022-10-10","arxiv_id":"2210.05372","n_code_links":1,"syntology":null},{"paper":"/paper/empowering-the-fact-checkers-automatic","slug":"empowering-the-fact-checkers-automatic","title":"Empowering the Fact-checkers! Automatic Identification of Claim Spans on Twitter","date":"2022-10-10","arxiv_id":"2210.04710","n_code_links":1,"syntology":null},{"paper":"/paper/multi-cls-bert-an-efficient-alternative-to","slug":"multi-cls-bert-an-efficient-alternative-to","title":"Multi-CLS BERT: An Efficient Alternative to Traditional Ensembling","date":"2022-10-10","arxiv_id":"2210.05043","n_code_links":1,"syntology":null},{"paper":"/paper/multiagent-reinforcement-learning-based-on","slug":"multiagent-reinforcement-learning-based-on","title":"Multiagent Reinforcement Learning Based on Fusion-Multiactor-Attention-Critic for Multiple-Unmanned-Aerial-Vehicle Navigation Control","date":"2022-10-10","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/rev-information-theoretic-evaluation-of-free","slug":"rev-information-theoretic-evaluation-of-free","title":"REV: Information-Theoretic Evaluation of Free-Text Rationales","date":"2022-10-10","arxiv_id":"2210.04982","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hanjiechen/rev"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-minimum-wage-as-an-anchor-effects-on","title":"The Minimum Wage as an Anchor: Effects on Determinations of Fairness by Humans and AI","date":"2022-10-10","arxiv_id":"2210.10585","n_code_links":0,"syntology":null},{"paper":"/paper/uncertainty-quantification-with-pre-trained","slug":"uncertainty-quantification-with-pre-trained","title":"Uncertainty Quantification with Pre-trained Language Models: A Large-Scale Empirical Analysis","date":"2022-10-10","arxiv_id":"2210.04714","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["xiaoyuxin1002/uq-plm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/asdot-any-shot-data-to-text-generation-with","slug":"asdot-any-shot-data-to-text-generation-with","title":"ASDOT: Any-Shot Data-to-Text Generation with Pretrained Language Models","date":"2022-10-09","arxiv_id":"2210.04325","n_code_links":1,"syntology":null},{"paper":"/paper/controllable-dialogue-simulation-with-in","slug":"controllable-dialogue-simulation-with-in","title":"Controllable Dialogue Simulation with In-Context Learning","date":"2022-10-09","arxiv_id":"2210.04185","n_code_links":1,"syntology":null},{"paper":"/paper/fairger-using-nlp-to-measure-support-for","slug":"fairger-using-nlp-to-measure-support-for","title":"Fine-Grained Detection of Solidarity for Women and Migrants in 155 Years of German Parliamentary Debates","date":"2022-10-09","arxiv_id":"2210.04359","n_code_links":2,"syntology":null},{"paper":"/paper/fine-tuning-pre-trained-transformers-into","slug":"fine-tuning-pre-trained-transformers-into","title":"Fine-Tuning Pre-trained Transformers into Decaying Fast Weights","date":"2022-10-09","arxiv_id":"2210.04243","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jenni-ai/t2fw"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improve-transformer-pre-training-with","title":"Better Pre-Training by Reducing Representation Confusion","date":"2022-10-09","arxiv_id":"2210.04246","n_code_links":0,"syntology":null},{"paper":"/paper/spread-love-not-hate-undermining-the","slug":"spread-love-not-hate-undermining-the","title":"Spread Love Not Hate: Undermining the Importance of Hateful Pre-training for Hate Speech Detection","date":"2022-10-09","arxiv_id":"2210.04267","n_code_links":1,"syntology":null},{"paper":null,"slug":"alphatuning-quantization-aware-parameter","title":"AlphaTuning: Quantization-Aware Parameter-Efficient Adaptation of Large-Scale Pre-Trained Language Models","date":"2022-10-08","arxiv_id":"2210.03858","n_code_links":0,"syntology":null},{"paper":null,"slug":"kg-mtt-bert-knowledge-graph-enhanced-bert-for","title":"KG-MTT-BERT: Knowledge Graph Enhanced BERT for Multi-Type Medical Text Classification","date":"2022-10-08","arxiv_id":"2210.03970","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-task-adaptive-pretraining-for-dialogue","title":"On Task-Adaptive Pretraining for Dialogue Response Selection","date":"2022-10-08","arxiv_id":"2210.04073","n_code_links":0,"syntology":null},{"paper":null,"slug":"algorithmic-trading-using-continuous-action","title":"Algorithmic Trading Using Continuous Action Space Deep Reinforcement Learning","date":"2022-10-07","arxiv_id":"2210.03469","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-chain-of-thought-prompting-in-large","slug":"automatic-chain-of-thought-prompting-in-large","title":"Automatic Chain of Thought Prompting in Large Language Models","date":"2022-10-07","arxiv_id":"2210.03493","n_code_links":5,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["amazon-science/auto-cot"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"paper":null,"slug":"dabert-dual-attention-enhanced-bert-for","title":"DABERT: Dual Attention Enhanced BERT for Semantic Matching","date":"2022-10-07","arxiv_id":"2210.03454","n_code_links":0,"syntology":null},{"paper":null,"slug":"generating-quizzes-to-support-training-on","title":"Generating Quizzes to Support Training on Quality Management and Assurance in Space Science and Engineering","date":"2022-10-07","arxiv_id":"2210.03427","n_code_links":0,"syntology":null},{"paper":"/paper/how-large-language-models-are-transforming","slug":"how-large-language-models-are-transforming","title":"How Large Language Models are Transforming Machine-Paraphrased Plagiarism","date":"2022-10-07","arxiv_id":"2210.03568","n_code_links":3,"syntology":null},{"paper":"/paper/knowledge-injected-prompt-based-fine-tuning","slug":"knowledge-injected-prompt-based-fine-tuning","title":"Knowledge Injected Prompt Based Fine-tuning for Multi-label Few-shot ICD Coding","date":"2022-10-07","arxiv_id":"2210.03304","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["whaleloops/KEPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/measuring-and-narrowing-the-compositionality","slug":"measuring-and-narrowing-the-compositionality","title":"Measuring and Narrowing the Compositionality Gap in Language Models","date":"2022-10-07","arxiv_id":"2210.03350","n_code_links":1,"syntology":null},{"paper":"/paper/uu-tax-at-semeval-2022-task-3-improving-the-1","slug":"uu-tax-at-semeval-2022-task-3-improving-the-1","title":"UU-Tax at SemEval-2022 Task 3: Improving the generalizability of language models for taxonomy classification through data augmentation","date":"2022-10-07","arxiv_id":"2210.03378","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-better-way-to-decay-proximal-gradient","title":"PathProx: A Proximal Gradient Algorithm for Weight Decay Regularized Deep Neural Networks","date":"2022-10-06","arxiv_id":"2210.03069","n_code_links":0,"syntology":null},{"paper":"/paper/binding-language-models-in-symbolic-languages","slug":"binding-language-models-in-symbolic-languages","title":"Binding Language Models in Symbolic Languages","date":"2022-10-06","arxiv_id":"2210.02875","n_code_links":4,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hkunlp/binder"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/bytetransformer-a-high-performance","slug":"bytetransformer-a-high-performance","title":"ByteTransformer: A High-Performance Transformer Boosted for Variable-Length Inputs","date":"2022-10-06","arxiv_id":"2210.03052","n_code_links":1,"syntology":null},{"paper":null,"slug":"explainable-verbal-deception-detection-using","title":"Explainable Verbal Deception Detection using Transformers","date":"2022-10-06","arxiv_id":"2210.03080","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalization-properties-of-retrieval-based","title":"Generalization Properties of Retrieval-based Models","date":"2022-10-06","arxiv_id":"2210.02617","n_code_links":0,"syntology":null},{"paper":"/paper/guess-the-instruction-making-language-models","slug":"guess-the-instruction-making-language-models","title":"Guess the Instruction! Flipped Learning Makes Language Models Stronger Zero-Shot Learners","date":"2022-10-06","arxiv_id":"2210.02969","n_code_links":1,"syntology":null},{"paper":"/paper/improving-the-domain-adaptation-of-retrieval","slug":"improving-the-domain-adaptation-of-retrieval","title":"Improving the Domain Adaptation of Retrieval Augmented Generation (RAG) Models for Open Domain Question Answering","date":"2022-10-06","arxiv_id":"2210.02627","n_code_links":1,"syntology":null},{"paper":null,"slug":"join-chain-network-a-logical-reasoning-view","title":"Join-Chain Network: A Logical Reasoning View of the Multi-head Attention in Transformer","date":"2022-10-06","arxiv_id":"2210.02729","n_code_links":0,"syntology":null},{"paper":null,"slug":"matching-text-and-audio-embeddings-exploring","title":"Matching Text and Audio Embeddings: Exploring Transfer-learning Strategies for Language-based Audio Retrieval","date":"2022-10-06","arxiv_id":"2210.02833","n_code_links":0,"syntology":null},{"paper":null,"slug":"murag-multimodal-retrieval-augmented","title":"MuRAG: Multimodal Retrieval-Augmented Generator for Open Question Answering over Images and Text","date":"2022-10-06","arxiv_id":"2210.02928","n_code_links":0,"syntology":null},{"paper":"/paper/rainier-reinforced-knowledge-introspector-for","slug":"rainier-reinforced-knowledge-introspector-for","title":"Rainier: Reinforced Knowledge Introspector for Commonsense Question Answering","date":"2022-10-06","arxiv_id":"2210.03078","n_code_links":1,"syntology":{"ran":7,"of":13,"n_ran_checked":4,"n_instrument":3,"unverified":6,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","official":{"repos":["liujch1998/rainier"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/glm-130b-an-open-bilingual-pre-trained-model","slug":"glm-130b-an-open-bilingual-pre-trained-model","title":"GLM-130B: An Open Bilingual Pre-trained Model","date":"2022-10-05","arxiv_id":"2210.02414","n_code_links":9,"syntology":{"ran":15,"of":21,"n_ran_checked":14,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["thudm/glm-130b"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"privacy-preserving-text-classification-on-1","title":"Privacy-Preserving Text Classification on BERT Embeddings with Homomorphic Encryption","date":"2022-10-05","arxiv_id":"2210.02574","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-structured-dropout","title":"Revisiting Structured Dropout","date":"2022-10-05","arxiv_id":"2210.02570","n_code_links":0,"syntology":null},{"paper":"/paper/explaining-patterns-in-data-with-language","slug":"explaining-patterns-in-data-with-language","title":"Explaining Patterns in Data with Language Models via Interpretable Autoprompting","date":"2022-10-04","arxiv_id":"2210.01848","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["csinva/imodelsX"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"characterization-of-effects-of-transfer","title":"The (In)Effectiveness of Intermediate Task Training For Domain Adaptation and Cross-Lingual Transfer Learning","date":"2022-10-03","arxiv_id":"2210.01091","n_code_links":0,"syntology":null},{"paper":null,"slug":"complexity-based-prompting-for-multi-step","title":"Complexity-Based Prompting for Multi-Step Reasoning","date":"2022-10-03","arxiv_id":"2210.00720","n_code_links":0,"syntology":null},{"paper":"/paper/language-models-are-greedy-reasoners-a","slug":"language-models-are-greedy-reasoners-a","title":"Language Models Are Greedy Reasoners: A Systematic Formal Analysis of Chain-of-Thought","date":"2022-10-03","arxiv_id":"2210.01240","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["asaparov/prontoqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/omnigrok-grokking-beyond-algorithmic-data","slug":"omnigrok-grokking-beyond-algorithmic-data","title":"Omnigrok: Grokking Beyond Algorithmic Data","date":"2022-10-03","arxiv_id":"2210.01117","n_code_links":2,"syntology":null},{"paper":null,"slug":"probing-of-quantitative-values-in-abstractive","title":"Probing of Quantitative Values in Abstractive Summarization Models","date":"2022-10-03","arxiv_id":"2210.00667","n_code_links":0,"syntology":null},{"paper":null,"slug":"construction-and-evaluation-of-a-self","title":"Construction and Evaluation of a Self-Attention Model for Semantic Understanding of Sentence-Final Particles","date":"2022-10-01","arxiv_id":"2210.00282","n_code_links":0,"syntology":null},{"paper":"/paper/promptkg-a-prompt-learning-framework-for","slug":"promptkg-a-prompt-learning-framework-for","title":"LambdaKG: A Library for Pre-trained Language Model-Based Knowledge Graph Embeddings","date":"2022-10-01","arxiv_id":"2210.00305","n_code_links":2,"syntology":null},{"paper":"/paper/adaptive-weight-decay-on-the-fly-weight-decay","slug":"adaptive-weight-decay-on-the-fly-weight-decay","title":"Improving Robustness with Adaptive Weight Decay","date":"2022-09-30","arxiv_id":"2210.00094","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/exploiting-selection-bias-on-underspecified","slug":"exploiting-selection-bias-on-underspecified","title":"Underspecification in Language Modeling Tasks: A Causality-Informed Study of Gendered Pronoun Resolution","date":"2022-09-30","arxiv_id":"2210.00131","n_code_links":2,"syntology":null},{"paper":"/paper/relu-neural-networks-learn-the-simplest","slug":"relu-neural-networks-learn-the-simplest","title":"Overparameterized ReLU Neural Networks Learn the Simplest Models: Neural Isometry and Exact Recovery","date":"2022-09-30","arxiv_id":"2209.15265","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pilancilab/neural-recovery"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scale-invariant-bayesian-neural-networks-with","title":"Scale-invariant Bayesian Neural Networks with Connectivity Tangent Kernel","date":"2022-09-30","arxiv_id":"2209.15208","n_code_links":0,"syntology":null},{"paper":"/paper/smallcap-lightweight-image-captioning","slug":"smallcap-lightweight-image-captioning","title":"SmallCap: Lightweight Image Captioning Prompted with Retrieval Augmentation","date":"2022-09-30","arxiv_id":"2209.15323","n_code_links":1,"syntology":null},{"paper":null,"slug":"bidirectional-language-models-are-also-few","title":"Bidirectional Language Models Are Also Few-shot Learners","date":"2022-09-29","arxiv_id":"2209.14500","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-prompt-learning-via-policy-gradient","slug":"dynamic-prompt-learning-via-policy-gradient","title":"Dynamic Prompt Learning via Policy Gradient for Semi-structured Mathematical Reasoning","date":"2022-09-29","arxiv_id":"2209.14610","n_code_links":2,"syntology":{"ran":3,"of":7,"n_ran_checked":1,"n_instrument":2,"unverified":4,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/nag-gs-semi-implicit-accelerated-and-robust","slug":"nag-gs-semi-implicit-accelerated-and-robust","title":"NAG-GS: Semi-Implicit, Accelerated and Robust Stochastic Optimizer","date":"2022-09-29","arxiv_id":"2209.14937","n_code_links":2,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["skolai/nag-gs","naggsopt/naggs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"neural-networks-efficiently-learn-low","title":"Neural Networks Efficiently Learn Low-Dimensional Representations with SGD","date":"2022-09-29","arxiv_id":"2209.14863","n_code_links":0,"syntology":null},{"paper":null,"slug":"cefer-a-four-facets-framework-based-on","title":"CEFER: A Four Facets Framework based on Context and Emotion embedded features for Implicit and Explicit Emotion Recognition","date":"2022-09-28","arxiv_id":"2209.13999","n_code_links":0,"syntology":null},{"paper":"/paper/downstream-datasets-make-surprisingly-good","slug":"downstream-datasets-make-surprisingly-good","title":"Downstream Datasets Make Surprisingly Good Pretraining Corpora","date":"2022-09-28","arxiv_id":"2209.14389","n_code_links":1,"syntology":null},{"paper":null,"slug":"medical-image-captioning-via-generative","title":"Medical Image Captioning via Generative Pretrained Transformers","date":"2022-09-28","arxiv_id":"2209.13983","n_code_links":0,"syntology":null},{"paper":null,"slug":"supervised-contrastive-learning-as-multi","title":"Supervised Contrastive Learning as Multi-Objective Optimization for Fine-Tuning Large Pre-trained Language Models","date":"2022-09-28","arxiv_id":"2209.14161","n_code_links":0,"syntology":null},{"paper":"/paper/who-is-gpt-3-an-exploration-of-personality","slug":"who-is-gpt-3-an-exploration-of-personality","title":"Who is GPT-3? An Exploration of Personality, Values and Demographics","date":"2022-09-28","arxiv_id":"2209.14338","n_code_links":1,"syntology":null},{"paper":"/paper/yato-yet-another-deep-learning-based-text","slug":"yato-yet-another-deep-learning-based-text","title":"YATO: Yet Another deep learning based Text analysis Open toolkit","date":"2022-09-28","arxiv_id":"2209.13877","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-critical-appraisal-of-equity-in","title":"How GPT-3 responds to different publics on climate change and Black Lives Matter: A critical appraisal of equity in conversational AI","date":"2022-09-27","arxiv_id":"2209.13627","n_code_links":0,"syntology":null},{"paper":null,"slug":"extractive-question-answering-on-queries-in","title":"Extractive Question Answering on Queries in Hindi and Tamil","date":"2022-09-27","arxiv_id":"2210.06356","n_code_links":0,"syntology":null},{"paper":"/paper/outlier-suppression-pushing-the-limit-of-low","slug":"outlier-suppression-pushing-the-limit-of-low","title":"Outlier Suppression: Pushing the Limit of Low-bit Transformer Language Models","date":"2022-09-27","arxiv_id":"2209.13325","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wimh966/outlier_suppression"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/wikides-a-wikipedia-based-dataset-for","slug":"wikides-a-wikipedia-based-dataset-for","title":"WikiDes: A Wikipedia-Based Dataset for Generating Short Descriptions from Paragraphs","date":"2022-09-27","arxiv_id":"2209.13101","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-ever-larger-octopi-still-amplify-reporting","title":"Do ever larger octopi still amplify reporting biases? Evidence from judgments of typical colour","date":"2022-09-26","arxiv_id":"2209.12786","n_code_links":0,"syntology":null},{"paper":"/paper/news-summarization-and-evaluation-in-the-era","slug":"news-summarization-and-evaluation-in-the-era","title":"News Summarization and Evaluation in the Era of GPT-3","date":"2022-09-26","arxiv_id":"2209.12356","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-simple-and-efficient-task-adaptive","title":"Towards Simple and Efficient Task-Adaptive Pre-training for Text Classification","date":"2022-09-26","arxiv_id":"2209.12943","n_code_links":0,"syntology":null},{"paper":null,"slug":"bigger-faster-two-stage-neural-architecture","title":"SpeedLimit: Neural Architecture Search for Quantized Transformer Models","date":"2022-09-25","arxiv_id":"2209.12127","n_code_links":0,"syntology":null},{"paper":null,"slug":"sentiment-analysis-on-inflation-after-covid","title":"Sentiment Analysis on Inflation after Covid-19","date":"2022-09-25","arxiv_id":"2209.14737","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-transformer-models-effectively-detect","title":"Can Transformer Models Effectively Detect Software Aspects in StackOverflow Discussion?","date":"2022-09-24","arxiv_id":"2209.12065","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-chess-with-language-models-and","title":"Learning Chess With Language Models and Transformers","date":"2022-09-24","arxiv_id":"2209.11902","n_code_links":0,"syntology":null},{"paper":null,"slug":"moral-mimicry-large-language-models-produce","title":"Moral Mimicry: Large Language Models Produce Moral Rationalizations Tailored to Political Identity","date":"2022-09-24","arxiv_id":"2209.12106","n_code_links":0,"syntology":null},{"paper":null,"slug":"idea-interactive-double-attentions-from-label","title":"IDEA: Interactive DoublE Attentions from Label Embedding for Text Classification","date":"2022-09-23","arxiv_id":"2209.11407","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-case-report-on-the-a-i-locked-in-problem","title":"A Case Report On The \"A.I. Locked-In Problem\": social concerns with modern NLP","date":"2022-09-22","arxiv_id":"2209.12687","n_code_links":0,"syntology":null},{"paper":"/paper/adaptation-of-domain-specific-transformer","slug":"adaptation-of-domain-specific-transformer","title":"Adaptation of domain-specific transformer models with text oversampling for sentiment analysis of social media posts on Covid-19 vaccines","date":"2022-09-22","arxiv_id":"2209.10966","n_code_links":1,"syntology":null},{"paper":null,"slug":"air-jpmc-smm4h-22-classifying-self-reported","title":"AIR-JPMC@SMM4H'22: Classifying Self-Reported Intimate Partner Violence in Tweets with Multiple BERT-based Models","date":"2022-09-22","arxiv_id":"2209.10763","n_code_links":0,"syntology":null},{"paper":null,"slug":"dfx-a-low-latency-multi-fpga-appliance-for","title":"DFX: A Low-latency Multi-FPGA Appliance for Accelerating Transformer-based Text Generation","date":"2022-09-22","arxiv_id":"2209.10797","n_code_links":0,"syntology":null},{"paper":null,"slug":"minimizing-human-assistance-augmenting-a","title":"Minimizing Human Assistance: Augmenting a Single Demonstration for Deep Reinforcement Learning","date":"2022-09-22","arxiv_id":"2209.11275","n_code_links":0,"syntology":null},{"paper":"/paper/bias-at-a-second-glance-a-deep-dive-into-bias","slug":"bias-at-a-second-glance-a-deep-dive-into-bias","title":"Bias at a Second Glance: A Deep Dive into Bias for German Educational Peer-Review Data Modeling","date":"2022-09-21","arxiv_id":"2209.10335","n_code_links":2,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["epfl-ml4ed/bias-at-a-second-glance"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["listed"]}}},{"paper":"/paper/cae-mechanism-to-diminish-the-class","slug":"cae-mechanism-to-diminish-the-class","title":"CAE: Mechanism to Diminish the Class Imbalanced in SLU Slot Filling Task","date":"2022-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"representing-affect-information-in-word","title":"Representing Affect Information in Word Embeddings","date":"2022-09-21","arxiv_id":"2209.10583","n_code_links":0,"syntology":null},{"paper":null,"slug":"subject-verb-agreement-error-patterns-in","title":"Subject Verb Agreement Error Patterns in Meaningless Sentences: Humans vs. BERT","date":"2022-09-21","arxiv_id":"2209.10538","n_code_links":0,"syntology":null},{"paper":null,"slug":"text-revealer-private-text-reconstruction-via","title":"Text Revealer: Private Text Reconstruction via Model Inversion Attacks against Transformers","date":"2022-09-21","arxiv_id":"2209.10505","n_code_links":0,"syntology":null},{"paper":null,"slug":"integer-fine-tuning-of-transformer-based","title":"Towards Fine-tuning Pre-trained Language Models with Integer Forward and Backward Propagation","date":"2022-09-20","arxiv_id":"2209.09815","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-to-many-semantic-communication-systems","title":"One-to-Many Semantic Communication Systems: Design, Implementation, Performance Evaluation","date":"2022-09-20","arxiv_id":"2209.09425","n_code_links":0,"syntology":null}],"record_sha256":"9bcb2b0a831c76fd813d5341f8f6d099ddb481d226fa64cba4549847ef8bf01b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}