{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-cosine-annealing/papers/7","list_of":"/method/linear-warmup-with-cosine-annealing","method":"Linear Warmup With Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":38,"rows_per_page":100,"rows":[601,700],"of":3797,"counts":{"archive_papers_tagged":3797,"with_a_code_link":1655,"where_syntology_ran_a_sample":602,"not_listed_spam_title":0,"listed":3797,"listed_where_code_ran":602,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":490,"every_run_a_failure_of_syntologys_instrument":112,"listed_with_a_run_with_no_instrument_failure":490,"listed_every_run_a_failure_of_syntologys_instrument":112,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-cosine-annealing","prev":"/method/linear-warmup-with-cosine-annealing/papers/6","next":"/method/linear-warmup-with-cosine-annealing/papers/8","papers":[{"paper":"/paper/rethinking-legal-judgement-prediction-in-a","slug":"rethinking-legal-judgement-prediction-in-a","title":"Rethinking Legal Judgement Prediction in a Realistic Scenario in the Era of Large Language Models","date":"2024-10-14","arxiv_id":"2410.10542","n_code_links":1,"syntology":null},{"paper":"/paper/towards-better-multi-head-attention-via","slug":"towards-better-multi-head-attention-via","title":"Towards Better Multi-head Attention via Channel-wise Sample Permutation","date":"2024-10-14","arxiv_id":"2410.10914","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dashenzi721/csp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-in-context-learning-really-generalize-to","title":"Can In-context Learning Really Generalize to Out-of-distribution Tasks?","date":"2024-10-13","arxiv_id":"2410.09695","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-gender-bias-of-llms-in-making","title":"Evaluating Gender Bias of LLMs in Making Morality Judgements","date":"2024-10-13","arxiv_id":"2410.09992","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-implicit-bias-in-large-language","title":"Investigating Implicit Bias in Large Language Models: A Large-Scale Study of Over 50 LLMs","date":"2024-10-13","arxiv_id":"2410.12864","n_code_links":0,"syntology":null},{"paper":null,"slug":"m2m-gen-a-multimodal-framework-for-automated","title":"M2M-Gen: A Multimodal Framework for Automated Background Music Generation in Japanese Manga Using Large Language Models","date":"2024-10-13","arxiv_id":"2410.09928","n_code_links":0,"syntology":null},{"paper":null,"slug":"llinstruct-an-instruction-tuned-model-for","title":"\\llinstruct: An Instruction-tuned model for English Language Proficiency Assessments","date":"2024-10-12","arxiv_id":"2410.09314","n_code_links":0,"syntology":null},{"paper":"/paper/attngcg-enhancing-jailbreaking-attacks-on","slug":"attngcg-enhancing-jailbreaking-attacks-on","title":"AttnGCG: Enhancing Jailbreaking Attacks on LLMs with Attention Manipulation","date":"2024-10-11","arxiv_id":"2410.09040","n_code_links":1,"syntology":null},{"paper":null,"slug":"extra-global-attention-designation-using","title":"Extra Global Attention Designation Using Keyword Detection in Sparse Transformer Architectures","date":"2024-10-11","arxiv_id":"2410.08971","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-in-house-large-language-models-to","title":"Fine-Tuning In-House Large Language Models to Infer Differential Diagnosis from Radiology Reports","date":"2024-10-11","arxiv_id":"2410.09234","n_code_links":0,"syntology":null},{"paper":null,"slug":"humanity-in-ai-detecting-the-personality-of","title":"Humanity in AI: Detecting the Personality of Large Language Models","date":"2024-10-11","arxiv_id":"2410.08545","n_code_links":0,"syntology":null},{"paper":null,"slug":"observing-the-southern-us-culture-of-honor","title":"Observing the Southern US Culture of Honor Using Large-Scale Social Media Analysis","date":"2024-10-11","arxiv_id":"2410.13887","n_code_links":0,"syntology":null},{"paper":"/paper/socialgaze-improving-the-integration-of-human","slug":"socialgaze-improving-the-integration-of-human","title":"SocialGaze: Improving the Integration of Human Social Norms in Large Language Models","date":"2024-10-11","arxiv_id":"2410.08698","n_code_links":1,"syntology":null},{"paper":"/paper/synth-sonar-sonar-image-synthesis-with","slug":"synth-sonar-sonar-image-synthesis-with","title":"Synth-SONAR: Sonar Image Synthesis with Enhanced Diversity and Realism via Dual Diffusion Models and GPT Prompting","date":"2024-10-11","arxiv_id":"2410.08612","n_code_links":1,"syntology":null},{"paper":"/paper/adam-exploits-ell-infty-geometry-of-loss","slug":"adam-exploits-ell-infty-geometry-of-loss","title":"Adam Exploits $\\ell_\\infty$-geometry of Loss Landscape via Coordinate-wise Adaptivity","date":"2024-10-10","arxiv_id":"2410.08198","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mohamad-amin/adam-coordinate-adaptivity"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"flier-few-shot-language-image-models-embedded","title":"FLIER: Few-shot Language Image Models Embedded with Latent Representations","date":"2024-10-10","arxiv_id":"2410.07648","n_code_links":0,"syntology":null},{"paper":"/paper/the-rise-of-ai-generated-content-in-wikipedia","slug":"the-rise-of-ai-generated-content-in-wikipedia","title":"The Rise of AI-Generated Content in Wikipedia","date":"2024-10-10","arxiv_id":"2410.08044","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["brooksca3/wiki_collection"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"autofeedback-an-llm-based-framework-for","title":"AutoFeedback: An LLM-based Framework for Efficient and Accurate API Request Generation","date":"2024-10-09","arxiv_id":"2410.06943","n_code_links":0,"syntology":null},{"paper":null,"slug":"capturing-bias-diversity-in-llms","title":"Capturing Bias Diversity in LLMs","date":"2024-10-09","arxiv_id":"2410.12839","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-model-for-less-resourced-language","title":"Generative Model for Less-Resourced Language with 1 billion parameters","date":"2024-10-09","arxiv_id":"2410.06898","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-code-executors-an","title":"Large Language Models as Code Executors: An Exploratory Study","date":"2024-10-09","arxiv_id":"2410.06667","n_code_links":0,"syntology":null},{"paper":"/paper/mentalarena-self-play-training-of-language","slug":"mentalarena-self-play-training-of-language","title":"MentalArena: Self-play Training of Language Models for Diagnosis and Treatment of Mental Health Disorders","date":"2024-10-09","arxiv_id":"2410.06845","n_code_links":1,"syntology":null},{"paper":null,"slug":"sage-scalable-ground-truth-evaluations-for","title":"SAGE: Scalable Ground Truth Evaluations for Large Sparse Autoencoders","date":"2024-10-09","arxiv_id":"2410.07456","n_code_links":0,"syntology":null},{"paper":"/paper/a-second-order-like-optimizer-with-adaptive","slug":"a-second-order-like-optimizer-with-adaptive","title":"A second-order-like optimizer with adaptive gradient scaling for deep learning","date":"2024-10-08","arxiv_id":"2410.05871","n_code_links":1,"syntology":null},{"paper":null,"slug":"auto-evolve-enhancing-large-language-model-s","title":"Auto-Evolve: Enhancing Large Language Model's Performance via Self-Reasoning Framework","date":"2024-10-08","arxiv_id":"2410.06328","n_code_links":0,"syntology":null},{"paper":"/paper/coevolving-with-the-other-you-fine-tuning-llm","slug":"coevolving-with-the-other-you-fine-tuning-llm","title":"Coevolving with the Other You: Fine-Tuning LLM with Sequential Cooperative Multi-Agent Reinforcement Learning","date":"2024-10-08","arxiv_id":"2410.06101","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":6,"n_instrument":2,"unverified":3,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Harry67Hu/CORY"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"leveraging-free-energy-in-pretraining-model","title":"Leveraging free energy in pretraining model selection for improved fine-tuning","date":"2024-10-08","arxiv_id":"2410.05612","n_code_links":0,"syntology":null},{"paper":null,"slug":"anyattack-towards-large-scale-self-supervised","title":"AnyAttack: Towards Large-scale Self-supervised Adversarial Attacks on Vision-language Models","date":"2024-10-07","arxiv_id":"2410.05346","n_code_links":0,"syntology":null},{"paper":null,"slug":"lpzero-language-model-zero-cost-proxy-search","title":"LPZero: Language Model Zero-cost Proxy Search from Zero","date":"2024-10-07","arxiv_id":"2410.04808","n_code_links":0,"syntology":null},{"paper":"/paper/narrative-of-thought-improving-temporal","slug":"narrative-of-thought-improving-temporal","title":"Narrative-of-Thought: Improving Temporal Reasoning of Large Language Models via Recounted Narratives","date":"2024-10-07","arxiv_id":"2410.05558","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-instruction-finetuning-neural-machine","title":"On Instruction-Finetuning Neural Machine Translation Models","date":"2024-10-07","arxiv_id":"2410.05553","n_code_links":0,"syntology":null},{"paper":"/paper/famma-a-benchmark-for-financial-domain","slug":"famma-a-benchmark-for-financial-domain","title":"FAMMA: A Benchmark for Financial Domain Multilingual Multimodal Question Answering","date":"2024-10-06","arxiv_id":"2410.04526","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["famma-bench/bench-script"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-model-inference-acceleration-a","slug":"large-language-model-inference-acceleration-a","title":"Large Language Model Inference Acceleration: A Comprehensive Hardware Perspective","date":"2024-10-06","arxiv_id":"2410.04466","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-for-knowledge-free","title":"Large Language Models for Knowledge-Free Network Management: Feasibility Study and Opportunities","date":"2024-10-06","arxiv_id":"2410.17259","n_code_links":0,"syntology":null},{"paper":null,"slug":"protocollm-automatic-evaluation-framework-of","title":"ProtocoLLM: Automatic Evaluation Framework of LLMs on Domain-Specific Scientific Protocol Formulation Tasks","date":"2024-10-06","arxiv_id":"2410.04601","n_code_links":0,"syntology":null},{"paper":"/paper/gamified-crowd-sourcing-of-high-quality-data","slug":"gamified-crowd-sourcing-of-high-quality-data","title":"Gamified crowd-sourcing of high-quality data for visual fine-tuning","date":"2024-10-05","arxiv_id":"2410.04038","n_code_links":0,"syntology":null},{"paper":"/paper/take-it-easy-label-adaptive-self","slug":"take-it-easy-label-adaptive-self","title":"Take It Easy: Label-Adaptive Self-Rationalization for Fact Verification and Explanation Generation","date":"2024-10-05","arxiv_id":"2410.04002","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jingyng/label-adaptive-self-rationalization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"crafting-narrative-closures-zero-shot","title":"Crafting Narrative Closures: Zero-Shot Learning with SSM Mamba for Short Story Ending Generation","date":"2024-10-04","arxiv_id":"2410.10848","n_code_links":0,"syntology":null},{"paper":null,"slug":"cross-lingual-transfer-for-automatic-question","title":"Cross-lingual Transfer for Automatic Question Generation by Learning Interrogative Structures in Target Languages","date":"2024-10-04","arxiv_id":"2410.03197","n_code_links":0,"syntology":null},{"paper":"/paper/how-language-models-prioritize-contextual","slug":"how-language-models-prioritize-contextual","title":"How Language Models Prioritize Contextual Grammatical Cues?","date":"2024-10-04","arxiv_id":"2410.03447","n_code_links":1,"syntology":null},{"paper":"/paper/steering-large-language-models-between-code","slug":"steering-large-language-models-between-code","title":"Steering Large Language Models between Code Execution and Textual Reasoning","date":"2024-10-04","arxiv_id":"2410.03524","n_code_links":1,"syntology":null},{"paper":null,"slug":"structured-list-grounded-question-answering","title":"Structured List-Grounded Question Answering","date":"2024-10-04","arxiv_id":"2410.03950","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-linguistically-aware-and-language","title":"Towards Linguistically-Aware and Language-Independent Tokenization for Large Language Models (LLMs)","date":"2024-10-04","arxiv_id":"2410.03568","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-prompts-to-guide-large-language-models","title":"Using Prompts to Guide Large Language Models in Imitating a Real Person's Language Style","date":"2024-10-04","arxiv_id":"2410.03848","n_code_links":0,"syntology":null},{"paper":null,"slug":"alphaintegrator-transformer-action-search-for","title":"AlphaIntegrator: Transformer Action Search for Symbolic Integration Proofs","date":"2024-10-03","arxiv_id":"2410.02666","n_code_links":0,"syntology":null},{"paper":"/paper/codejudge-evaluating-code-generation-with","slug":"codejudge-evaluating-code-generation-with","title":"CodeJudge: Evaluating Code Generation with Large Language Models","date":"2024-10-03","arxiv_id":"2410.02184","n_code_links":1,"syntology":{"ran":15,"of":23,"n_ran_checked":14,"n_instrument":1,"unverified":8,"pointer_only":2,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 2 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["VichyTong/CodeJudge"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["found_in_text","official"]}}},{"paper":null,"slug":"llava-critic-learning-to-evaluate-multimodal","title":"LLaVA-Critic: Learning to Evaluate Multimodal Models","date":"2024-10-03","arxiv_id":"2410.02712","n_code_links":0,"syntology":null},{"paper":null,"slug":"plots-unlock-time-series-understanding-in","title":"Plots Unlock Time-Series Understanding in Multimodal Models","date":"2024-10-03","arxiv_id":"2410.02637","n_code_links":0,"syntology":null},{"paper":"/paper/visual-editing-with-llm-based-tool-chaining","slug":"visual-editing-with-llm-based-tool-chaining","title":"Visual Editing with LLM-based Tool Chaining: An Efficient Distillation Approach for Real-Time Applications","date":"2024-10-03","arxiv_id":"2410.02952","n_code_links":1,"syntology":null},{"paper":"/paper/automatic-deductive-coding-in-discourse","slug":"automatic-deductive-coding-in-discourse","title":"Automatic deductive coding in discourse analysis: an application of large language models in learning analytics","date":"2024-10-02","arxiv_id":"2410.01240","n_code_links":1,"syntology":null},{"paper":null,"slug":"emotion-aware-response-generation-using","title":"Emotion-Aware Embedding Fusion in LLMs (Flan-T5, LLAMA 2, DeepSeek-R1, and ChatGPT 4) for Intelligent Response Generation","date":"2024-10-02","arxiv_id":"2410.01306","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-llm-fine-tuning-for-text-to-sqls-by","title":"Enhancing LLM Fine-tuning for Text-to-SQLs by SQL Quality Measurement","date":"2024-10-02","arxiv_id":"2410.01869","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-adaptation-of-unlimiformer-for-decoder","title":"On The Adaptation of Unlimiformer for Decoder-Only Transformers","date":"2024-10-02","arxiv_id":"2410.01637","n_code_links":0,"syntology":null},{"paper":"/paper/quantifying-generalization-complexity-for","slug":"quantifying-generalization-complexity-for","title":"Quantifying Generalization Complexity for Large Language Models","date":"2024-10-02","arxiv_id":"2410.01769","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhentingqi/scylla"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/seeing-eye-to-ai-human-alignment-via-gaze","slug":"seeing-eye-to-ai-human-alignment-via-gaze","title":"Seeing Eye to AI: Human Alignment via Gaze-Based Response Rewards for Large Language Models","date":"2024-10-02","arxiv_id":"2410.01532","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["telefonica-scientific-research/gaze_reward"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/alignsum-data-pyramid-hierarchical-fine","slug":"alignsum-data-pyramid-hierarchical-fine","title":"AlignSum: Data Pyramid Hierarchical Fine-tuning for Aligning with Human Summarization Preference","date":"2024-10-01","arxiv_id":"2410.00409","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["csyanghan/alignsum"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"decoding-hate-exploring-language-models","title":"Decoding Hate: Exploring Language Models' Reactions to Hate Speech","date":"2024-10-01","arxiv_id":"2410.00775","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-enhanced-model-for-eye-leme-an-open","title":"Language Enhanced Model for Eye (LEME): An Open-Source Ophthalmology-Specific Large Language Model","date":"2024-10-01","arxiv_id":"2410.03740","n_code_links":0,"syntology":null},{"paper":"/paper/sparse-attention-decomposition-applied-to","slug":"sparse-attention-decomposition-applied-to","title":"Sparse Attention Decomposition Applied to Circuit Tracing","date":"2024-10-01","arxiv_id":"2410.00340","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-looming-replication-crisis-in-evaluating","title":"A Looming Replication Crisis in Evaluating Behavior in Language Models? Evidence and Solutions","date":"2024-09-30","arxiv_id":"2409.20303","n_code_links":0,"syntology":null},{"paper":null,"slug":"adapting-llms-for-the-medical-domain-in","title":"Adapting LLMs for the Medical Domain in Portuguese: A Study on Fine-Tuning and Model Evaluation","date":"2024-09-30","arxiv_id":"2410.00163","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-the-fairness-of-task-adaptive","slug":"evaluating-the-fairness-of-task-adaptive","title":"Evaluating the fairness of task-adaptive pretraining on unlabeled test data before few-shot text classification","date":"2024-09-30","arxiv_id":"2410.00179","n_code_links":1,"syntology":null},{"paper":null,"slug":"modelando-procesos-cognitivos-de-la-lectura","title":"Modelando procesos cognitivos de la lectura natural con GPT-2","date":"2024-09-30","arxiv_id":"2409.20174","n_code_links":0,"syntology":null},{"paper":"/paper/analog-in-memory-computing-attention","slug":"analog-in-memory-computing-attention","title":"Analog In-Memory Computing Attention Mechanism for Fast and Energy-Efficient Large Language Models","date":"2024-09-28","arxiv_id":"2409.19315","n_code_links":1,"syntology":null},{"paper":null,"slug":"charting-the-future-using-chart-question","title":"Charting the Future: Using Chart Question-Answering for Scalable Evaluation of LLM-Driven Data Visualizations","date":"2024-09-27","arxiv_id":"2409.18764","n_code_links":0,"syntology":null},{"paper":"/paper/cottention-linear-transformers-with-cosine","slug":"cottention-linear-transformers-with-cosine","title":"Cottention: Linear Transformers With Cosine Attention","date":"2024-09-27","arxiv_id":"2409.18747","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gmongaras/Cottention_Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"experimental-evaluation-of-machine-learning","title":"Experimental Evaluation of Machine Learning Models for Goal-oriented Customer Service Chatbot with Pipeline Architecture","date":"2024-09-27","arxiv_id":"2409.18568","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-unidirectional-bidirectional-and","title":"Comparing Unidirectional, Bidirectional, and Word2vec Models for Discovering Vulnerabilities in Compiled Lifted Code","date":"2024-09-26","arxiv_id":"2409.17513","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-in-domain-question-answering-for","title":"Efficient In-Domain Question Answering for Resource-Constrained Environments","date":"2024-09-26","arxiv_id":"2409.17648","n_code_links":0,"syntology":null},{"paper":"/paper/maskllm-learnable-semi-structured-sparsity","slug":"maskllm-learnable-semi-structured-sparsity","title":"MaskLLM: Learnable Semi-Structured Sparsity for Large Language Models","date":"2024-09-26","arxiv_id":"2409.17481","n_code_links":1,"syntology":{"ran":5,"of":16,"n_ran_checked":5,"n_instrument":0,"unverified":11,"pointer_only":16,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["nvlabs/maskllm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":11,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"t3-a-novel-zero-shot-transfer-learning","title":"T3: A Novel Zero-shot Transfer Learning Framework Iteratively Training on an Assistant Task for a Target Task","date":"2024-09-26","arxiv_id":"2409.17640","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-application-of-gpt-4-in-grading-design","title":"The application of GPT-4 in grading design university students' assignment and providing feedback: An exploratory study","date":"2024-09-26","arxiv_id":"2409.17698","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-prompting-based-representation-learning","title":"A Prompting-Based Representation Learning Method for Recommendation with Large Language Models","date":"2024-09-25","arxiv_id":"2409.16674","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-and-machine-learning-advancing","title":"Deep Learning and Machine Learning, Advancing Big Data Analytics and Management: Handy Appetizer","date":"2024-09-25","arxiv_id":"2409.17120","n_code_links":0,"syntology":null},{"paper":null,"slug":"severity-prediction-in-mental-health-llm","title":"Severity Prediction in Mental Health: LLM-based Creation, Analysis, Evaluation of a Novel Multilingual Dataset","date":"2024-09-25","arxiv_id":"2409.17397","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-llm-for-real-time-transcription-and","title":"Using LLM for Real-Time Transcription and Summarization of Doctor-Patient Interactions into ePuskesmas in Indonesia","date":"2024-09-25","arxiv_id":"2409.17054","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-can-be-cognitively-biased-an-exploratory","title":"AI Can Be Cognitively Biased: An Exploratory Study on Threshold Priming in LLM-Based Batch Relevance Assessment","date":"2024-09-24","arxiv_id":"2409.16022","n_code_links":0,"syntology":null},{"paper":"/paper/data-augmentation-for-sparse-multidimensional","slug":"data-augmentation-for-sparse-multidimensional","title":"Data Augmentation for Sparse Multidimensional Learning Performance Data Using Generative AI","date":"2024-09-24","arxiv_id":"2409.15631","n_code_links":1,"syntology":null},{"paper":"/paper/effectiveness-of-cross-linguistic-extraction","slug":"effectiveness-of-cross-linguistic-extraction","title":"Effectiveness of Cross-linguistic Extraction of Genetic Information using Generative Large Language Models","date":"2024-09-24","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"selection-of-prompt-engineering-techniques","title":"Selection of Prompt Engineering Techniques for Code Generation through Predicting Code Complexity","date":"2024-09-24","arxiv_id":"2409.16416","n_code_links":0,"syntology":null},{"paper":"/paper/self-attention-as-an-attractor-network","slug":"self-attention-as-an-attractor-network","title":"Self-attention as an attractor network: transient memories without backpropagation","date":"2024-09-24","arxiv_id":"2409.16112","n_code_links":1,"syntology":null},{"paper":null,"slug":"synatra-turning-indirect-knowledge-into","title":"Synatra: Turning Indirect Knowledge into Direct Demonstrations for Digital Agents at Scale","date":"2024-09-24","arxiv_id":"2409.15637","n_code_links":0,"syntology":null},{"paper":null,"slug":"task-oriented-prompt-enhancement-via-script","title":"Task-oriented Prompt Enhancement via Script Generation","date":"2024-09-24","arxiv_id":"2409.16418","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-depression-detection-on-social","title":"Advancing Depression Detection on Social Media Platforms Through Fine-Tuned Large Language Models","date":"2024-09-23","arxiv_id":"2409.14794","n_code_links":0,"syntology":null},{"paper":null,"slug":"chattronics-using-gpts-to-assist-in-the","title":"Chattronics: using GPTs to assist in the design of data acquisition systems","date":"2024-09-23","arxiv_id":"2409.15183","n_code_links":0,"syntology":null},{"paper":"/paper/effective-and-evasive-fuzz-testing-driven","slug":"effective-and-evasive-fuzz-testing-driven","title":"PAPILLON: Efficient and Stealthy Fuzz Testing-Powered Jailbreaks for LLMs","date":"2024-09-23","arxiv_id":"2409.14866","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aaFrostnova/Papillon"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gem-rag-graphical-eigen-memories-for","title":"GEM-RAG: Graphical Eigen Memories For Retrieval Augmented Generation","date":"2024-09-23","arxiv_id":"2409.15566","n_code_links":0,"syntology":null},{"paper":null,"slug":"location-is-key-leveraging-large-language","title":"Location is Key: Leveraging Large Language Model for Functional Bug Localization in Verilog","date":"2024-09-23","arxiv_id":"2409.15186","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-policy-analysis-through-prompt","title":"Privacy Policy Analysis through Prompt Engineering for LLMs","date":"2024-09-23","arxiv_id":"2409.14879","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-guard-an-llm-agent-for-real-time-voice","title":"Safe Guard: an LLM-agent for Real-time Voice-based Hate Speech Detection in Social Virtual Reality","date":"2024-09-23","arxiv_id":"2409.15623","n_code_links":0,"syntology":null},{"paper":"/paper/sdba-a-stealthy-and-long-lasting-durable","slug":"sdba-a-stealthy-and-long-lasting-durable","title":"SDBA: A Stealthy and Long-Lasting Durable Backdoor Attack in Federated Learning","date":"2024-09-23","arxiv_id":"2409.14805","n_code_links":1,"syntology":null},{"paper":"/paper/can-pre-trained-language-models-generate","slug":"can-pre-trained-language-models-generate","title":"Can pre-trained language models generate titles for research papers?","date":"2024-09-22","arxiv_id":"2409.14602","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-quality-of-code-comments","title":"Evaluating the Quality of Code Comments Generated by Large Language Models for Novice Programmers","date":"2024-09-22","arxiv_id":"2409.14368","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-are-one-shot-url-classifiers-and","title":"LLMs are One-Shot URL Classifiers and Explainers","date":"2024-09-22","arxiv_id":"2409.14306","n_code_links":0,"syntology":null},{"paper":null,"slug":"proof-automation-with-large-language-models","title":"Proof Automation with Large Language Models","date":"2024-09-22","arxiv_id":"2409.14274","n_code_links":0,"syntology":null},{"paper":"/paper/2409-14037","slug":"2409-14037","title":"Can LLMs replace Neil deGrasse Tyson? Evaluating the Reliability of LLMs as Science Communicators","date":"2024-09-21","arxiv_id":"2409.14037","n_code_links":1,"syntology":null},{"paper":"/paper/2409-14175","slug":"2409-14175","title":"QMOS: Enhancing LLMs for Telecommunication with Question Masked loss and Option Shuffling","date":"2024-09-21","arxiv_id":"2409.14175","n_code_links":1,"syntology":null},{"paper":null,"slug":"drift-to-remember","title":"Drift to Remember","date":"2024-09-21","arxiv_id":"2409.13997","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-in-triples-for-llms-enhancing-table","title":"Knowledge in Triples for LLMs: Enhancing Table QA Accuracy with Semantic Extraction","date":"2024-09-21","arxiv_id":"2409.14192","n_code_links":0,"syntology":null},{"paper":"/paper/loop-residual-neural-networks-for-iterative","slug":"loop-residual-neural-networks-for-iterative","title":"Loop Neural Networks for Parameter Sharing","date":"2024-09-21","arxiv_id":"2409.14199","n_code_links":0,"syntology":null}],"record_sha256":"dd977a4b5cde8816340b3b93f7d858b2968ebd64490138443b999f395d7c932b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}