{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-cosine-annealing/papers/17","list_of":"/method/linear-warmup-with-cosine-annealing","method":"Linear Warmup With Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":17,"pages_in_order":38,"rows_per_page":100,"rows":[1601,1700],"of":3797,"counts":{"archive_papers_tagged":3797,"with_a_code_link":1655,"where_syntology_ran_a_sample":602,"not_listed_spam_title":0,"listed":3797,"listed_where_code_ran":602,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":490,"every_run_a_failure_of_syntologys_instrument":112,"listed_with_a_run_with_no_instrument_failure":490,"listed_every_run_a_failure_of_syntologys_instrument":112,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-cosine-annealing","prev":"/method/linear-warmup-with-cosine-annealing/papers/16","next":"/method/linear-warmup-with-cosine-annealing/papers/18","papers":[{"paper":"/paper/autotimes-autoregressive-time-series","slug":"autotimes-autoregressive-time-series","title":"AutoTimes: Autoregressive Time Series Forecasters via Large Language Models","date":"2024-02-04","arxiv_id":"2402.02370","n_code_links":1,"syntology":null},{"paper":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","n_code_links":1,"syntology":{"ran":13,"of":18,"n_ran_checked":13,"n_instrument":0,"unverified":5,"pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-assessment-of-tutoring-practices","title":"Improving Assessment of Tutoring Practices using Retrieval-Augmented Generation","date":"2024-02-04","arxiv_id":"2402.14594","n_code_links":0,"syntology":null},{"paper":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tsis-a-supplementary-algorithm-to-t-smiles","slug":"tsis-a-supplementary-algorithm-to-t-smiles","title":"Hierarchical Structure Enhances the Convergence and Generalizability of Linear Molecular Representation","date":"2024-02-03","arxiv_id":"2402.02164","n_code_links":1,"syntology":null},{"paper":null,"slug":"comet-generating-commit-messages-using-delta","title":"COMET: Generating Commit Messages using Delta Graph Context Representation","date":"2024-02-02","arxiv_id":"2402.01841","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-the-limitations-of-graph-reasoning","slug":"exploring-the-limitations-of-graph-reasoning","title":"Can LLMs perform structured graph reasoning?","date":"2024-02-02","arxiv_id":"2402.01805","n_code_links":1,"syntology":null},{"paper":"/paper/improving-sequential-recommendations-with","slug":"improving-sequential-recommendations-with","title":"Improving Sequential Recommendations with LLMs","date":"2024-02-02","arxiv_id":"2402.01339","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dh-r/llm-sequential-recommendation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generation-distillation-and-evaluation-of","title":"Generation, Distillation and Evaluation of Motivational Interviewing-Style Reflections with a Foundational Language Model","date":"2024-02-01","arxiv_id":"2402.01051","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-planning-based-reasoning-by","title":"Learning Planning-based Reasoning by Trajectories Collection and Process Reward Synthesizing","date":"2024-02-01","arxiv_id":"2402.00658","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-contrastive-pre-training-for-1","title":"Self-Supervised Contrastive Pre-Training for Multivariate Point Processes","date":"2024-02-01","arxiv_id":"2402.00987","n_code_links":0,"syntology":null},{"paper":"/paper/sparql-generation-with-entity-pre-trained-gpt","slug":"sparql-generation-with-entity-pre-trained-gpt","title":"SPARQL Generation with Entity Pre-trained GPT for KG Question Answering","date":"2024-02-01","arxiv_id":"2402.00969","n_code_links":1,"syntology":null},{"paper":null,"slug":"tiny-titans-can-smaller-large-language-models","title":"Tiny Titans: Can Smaller Large Language Models Punch Above Their Weight in the Real World for Meeting Summarization?","date":"2024-02-01","arxiv_id":"2402.00841","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-scalable-robotic-intervention-of","title":"Human-mediated Large Language Models for Robotic Intervention in Children with Autism Spectrum Disorders","date":"2024-02-01","arxiv_id":"2402.00260","n_code_links":0,"syntology":null},{"paper":"/paper/consmax-hardware-friendly-alternative-softmax","slug":"consmax-hardware-friendly-alternative-softmax","title":"ConSmax: Hardware-Friendly Alternative Softmax with Learnable Parameters","date":"2024-01-31","arxiv_id":"2402.10930","n_code_links":1,"syntology":null},{"paper":null,"slug":"global-liar-factuality-of-llms-over-time-and","title":"Global-Liar: Factuality of LLMs over Time and Geographic Regions","date":"2024-01-31","arxiv_id":"2401.17839","n_code_links":0,"syntology":null},{"paper":null,"slug":"making-a-long-story-short-in-conversation","title":"Making a Long Story Short in Conversation Modeling","date":"2024-01-31","arxiv_id":"2402.00143","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-the-problem-of-strong-priors-in","title":"Mitigating the Influence of Distractor Tasks in LMs with Prior-Aware Decoding","date":"2024-01-31","arxiv_id":"2401.17692","n_code_links":0,"syntology":null},{"paper":null,"slug":"paramanu-a-family-of-novel-efficient-indic","title":"Paramanu: A Family of Novel Efficient Generative Foundation Language Models for Indian Languages","date":"2024-01-31","arxiv_id":"2401.18034","n_code_links":0,"syntology":null},{"paper":null,"slug":"real-sparks-of-artificial-intelligence-and","title":"Real Sparks of Artificial Intelligence and the Importance of Inner Interpretability","date":"2024-01-31","arxiv_id":"2402.00901","n_code_links":0,"syntology":null},{"paper":null,"slug":"uncertainty-aware-explainable-recommendation","title":"Uncertainty-Aware Explainable Recommendation with Large Language Models","date":"2024-01-31","arxiv_id":"2402.03366","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-preliminary-study-on-using-large-language","title":"A Preliminary Study on Using Large Language Models in Software Pentesting","date":"2024-01-30","arxiv_id":"2401.17459","n_code_links":0,"syntology":null},{"paper":"/paper/llamp-large-language-model-made-powerful-for","slug":"llamp-large-language-model-made-powerful-for","title":"LLaMP: Large Language Model Made Powerful for High-fidelity Materials Knowledge Retrieval and Distillation","date":"2024-01-30","arxiv_id":"2401.17244","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chiang-yuan/llamp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mt-eval-a-multi-turn-capabilities-evaluation","slug":"mt-eval-a-multi-turn-capabilities-evaluation","title":"MT-Eval: A Multi-Turn Capabilities Evaluation Benchmark for Large Language Models","date":"2024-01-30","arxiv_id":"2401.16745","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kwanwaichung/mt-eval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"diverse-but-divisive-llms-can-exaggerate","title":"Diverse, but Divisive: LLMs Can Exaggerate Gender Differences in Opinion Related to Harms of Misinformation","date":"2024-01-29","arxiv_id":"2401.16558","n_code_links":0,"syntology":null},{"paper":"/paper/e-eval-a-comprehensive-chinese-k-12-education","slug":"e-eval-a-comprehensive-chinese-k-12-education","title":"E-EVAL: A Comprehensive Chinese K-12 Education Evaluation Benchmark for Large Language Models","date":"2024-01-29","arxiv_id":"2401.15927","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ai-edu-lab/e-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"leveraging-professional-radiologists","title":"Leveraging Professional Radiologists' Expertise to Enhance LLMs' Evaluation for Radiology Reports","date":"2024-01-29","arxiv_id":"2401.16578","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm4vuln-a-unified-evaluation-framework-for","title":"LLM4Vuln: A Unified Evaluation Framework for Decoupling and Enhancing LLMs' Vulnerability Reasoning","date":"2024-01-29","arxiv_id":"2401.16185","n_code_links":0,"syntology":null},{"paper":"/paper/regal-refactoring-programs-to-discover","slug":"regal-refactoring-programs-to-discover","title":"ReGAL: Refactoring Programs to Discover Generalizable Abstractions","date":"2024-01-29","arxiv_id":"2401.16467","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["esteng/regal_program_learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"security-code-review-by-llms-a-deep-dive-into","title":"An Insight into Security Code Review with LLMs: Capabilities, Obstacles, and Influential Factors","date":"2024-01-29","arxiv_id":"2401.16310","n_code_links":0,"syntology":null},{"paper":null,"slug":"trackgpt-a-generative-pre-trained-transformer","title":"TrackGPT -- A generative pre-trained transformer for cross-domain entity trajectory forecasting","date":"2024-01-29","arxiv_id":"2402.00066","n_code_links":0,"syntology":null},{"paper":"/paper/convosense-overcoming-monotonous-commonsense","slug":"convosense-overcoming-monotonous-commonsense","title":"ConvoSense: Overcoming Monotonous Commonsense Inferences for Conversational AI","date":"2024-01-27","arxiv_id":"2401.15471","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-large-language-model-performance-to","title":"Enhancing Large Language Model Performance To Answer Questions and Extract Information More Accurately","date":"2024-01-27","arxiv_id":"2402.01722","n_code_links":0,"syntology":null},{"paper":null,"slug":"equipping-language-models-with-tool-use","title":"Equipping Language Models with Tool Use Capability for Tabular Data Analysis in Finance","date":"2024-01-27","arxiv_id":"2401.15328","n_code_links":0,"syntology":null},{"paper":null,"slug":"fortifying-ethical-boundaries-in-ai-advanced","title":"Fortifying Ethical Boundaries in AI: Advanced Strategies for Enhancing Security in Large Language Models","date":"2024-01-27","arxiv_id":"2402.01725","n_code_links":0,"syntology":null},{"paper":null,"slug":"geodecoder-empowering-multimodal-map","title":"GeoDecoder: Empowering Multimodal Map Understanding","date":"2024-01-26","arxiv_id":"2401.15118","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-qualitative-coding-with-llms-chain","title":"Scalable Qualitative Coding with LLMs: Chain-of-Thought Reasoning Matches Human Performance in Some Hermeneutic Tasks","date":"2024-01-26","arxiv_id":"2401.15170","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-zero-shot-inference","title":"A comparative study of zero-shot inference with large language models and supervised modeling in breast cancer pathology classification","date":"2024-01-25","arxiv_id":"2401.13887","n_code_links":0,"syntology":null},{"paper":"/paper/chat-gpt-v-bert-dawn-of-justice-for-semantic","slug":"chat-gpt-v-bert-dawn-of-justice-for-semantic","title":"(Chat)GPT v BERT: Dawn of Justice for Semantic Change Detection","date":"2024-01-25","arxiv_id":"2401.14040","n_code_links":1,"syntology":null},{"paper":"/paper/deepseek-coder-when-the-large-language-model","slug":"deepseek-coder-when-the-large-language-model","title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","date":"2024-01-25","arxiv_id":"2401.14196","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["deepseek-ai/DeepSeek-Coder"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-gpt-3-5-s-awareness-and","title":"Evaluating GPT-3.5's Awareness and Summarization Abilities for European Constitutional Texts with Shared Topics","date":"2024-01-25","arxiv_id":"2401.14524","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigate-consolidate-exploit-a-general","title":"Investigate-Consolidate-Exploit: A General Strategy for Inter-Task Agent Self-Evolution","date":"2024-01-25","arxiv_id":"2401.13996","n_code_links":0,"syntology":null},{"paper":"/paper/longhealth-a-question-answering-benchmark","slug":"longhealth-a-question-answering-benchmark","title":"LongHealth: A Question Answering Benchmark with Long Clinical Documents","date":"2024-01-25","arxiv_id":"2401.14490","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kbressem/longhealth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tricy-trigger-guided-data-to-text-generation-1","slug":"tricy-trigger-guided-data-to-text-generation-1","title":"TrICy: Trigger-guided Data-to-text Generation with Intent aware Attention-Copy","date":"2024-01-25","arxiv_id":"2402.01714","n_code_links":0,"syntology":null},{"paper":null,"slug":"unmasking-and-quantifying-racial-bias-of","title":"Unmasking and Quantifying Racial Bias of Large Language Models in Medical Report Generation","date":"2024-01-25","arxiv_id":"2401.13867","n_code_links":0,"syntology":null},{"paper":null,"slug":"zs4c-zero-shot-synthesis-of-compilable-code","title":"ZS4C: Zero-Shot Synthesis of Compilable Code for Incomplete Code Snippets using LLMs","date":"2024-01-25","arxiv_id":"2401.14279","n_code_links":0,"syntology":null},{"paper":"/paper/a-unified-approach-to-emotion-detection-and","slug":"a-unified-approach-to-emotion-detection-and","title":"A Unified Approach to Emotion Detection and Task-Oriented Dialogue Modeling","date":"2024-01-24","arxiv_id":"2401.13789","n_code_links":1,"syntology":null},{"paper":null,"slug":"automated-root-causing-of-cloud-incidents","title":"Automated Root Causing of Cloud Incidents using In-Context Learning with GPT-4","date":"2024-01-24","arxiv_id":"2401.13810","n_code_links":0,"syntology":null},{"paper":"/paper/can-gpt-3-5-generate-and-code-discharge","slug":"can-gpt-3-5-generate-and-code-discharge","title":"Can GPT-3.5 Generate and Code Discharge Summaries?","date":"2024-01-24","arxiv_id":"2401.13512","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["edinburghclinicalnlp/chatgpt_icd_coding"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"discovering-mathematical-formulas-from-data","title":"Discovering Mathematical Formulas from Data via GPT-guided Monte Carlo Tree Search","date":"2024-01-24","arxiv_id":"2401.14424","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-general-large-language-models","title":"Evaluation of General Large Language Models in Contextually Assessing Semantic Concepts Extracted from Adult Critical Care Electronic Health Record Notes","date":"2024-01-24","arxiv_id":"2401.13588","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-guided-question-answer-generation-for","title":"Graph Guided Question Answer Generation for Procedural Question-Answering","date":"2024-01-24","arxiv_id":"2401.13594","n_code_links":0,"syntology":null},{"paper":"/paper/how-good-is-chatgpt-at-face-biometrics-a","slug":"how-good-is-chatgpt-at-face-biometrics-a","title":"How Good is ChatGPT at Face Biometrics? A First Look into Recognition, Soft Biometrics, and Explainability","date":"2024-01-24","arxiv_id":"2401.13641","n_code_links":1,"syntology":null},{"paper":null,"slug":"segment-any-cell-a-sam-based-auto-prompting","title":"Segment Any Cell: A SAM-based Auto-prompting Fine-tuning Framework for Nuclei Segmentation","date":"2024-01-24","arxiv_id":"2401.13220","n_code_links":0,"syntology":null},{"paper":null,"slug":"kam-cot-knowledge-augmented-multimodal-chain","title":"KAM-CoT: Knowledge Augmented Multimodal Chain-of-Thoughts Reasoning","date":"2024-01-23","arxiv_id":"2401.12863","n_code_links":0,"syntology":null},{"paper":"/paper/trove-inducing-verifiable-and-efficient","slug":"trove-inducing-verifiable-and-efficient","title":"TroVE: Inducing Verifiable and Efficient Toolboxes for Solving Programmatic Tasks","date":"2024-01-23","arxiv_id":"2401.12869","n_code_links":1,"syntology":{"ran":16,"of":19,"n_ran_checked":16,"n_instrument":0,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zorazrw/trove"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-in-context-learning-via-linear","slug":"enhancing-in-context-learning-via-linear","title":"Enhancing In-context Learning via Linear Probe Calibration","date":"2024-01-22","arxiv_id":"2401.12406","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mominabbass/linc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/chex-gpt-harnessing-large-language-models-for","slug":"chex-gpt-harnessing-large-language-models-for","title":"CheX-GPT: Harnessing Large Language Models for Enhanced Chest X-ray Report Labeling","date":"2024-01-21","arxiv_id":"2401.11505","n_code_links":2,"syntology":null},{"paper":null,"slug":"enhancing-recommendation-diversity-by-re","title":"Enhancing Recommendation Diversity by Re-ranking with Large Language Models","date":"2024-01-21","arxiv_id":"2401.11506","n_code_links":0,"syntology":null},{"paper":"/paper/badchain-backdoor-chain-of-thought-prompting","slug":"badchain-backdoor-chain-of-thought-prompting","title":"BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models","date":"2024-01-20","arxiv_id":"2401.12242","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["django-jiang/badchain"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-large-language-models-for-clinical","title":"Enhancing Large Language Models for Clinical Decision Support by Incorporating Clinical Practice Guidelines","date":"2024-01-20","arxiv_id":"2401.11120","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-and-enhancing-large-language","title":"Evaluating and Enhancing Large Language Models Performance in Domain-specific Medicine: Osteoarthritis Management with DocOA","date":"2024-01-20","arxiv_id":"2401.12998","n_code_links":0,"syntology":null},{"paper":null,"slug":"finllms-a-framework-for-financial-reasoning","title":"FinLLMs: A Framework for Financial Reasoning Dataset Generation with Large Language Models","date":"2024-01-19","arxiv_id":"2401.10744","n_code_links":0,"syntology":null},{"paper":"/paper/mining-experimental-data-from-materials","slug":"mining-experimental-data-from-materials","title":"Mining experimental data from Materials Science literature with Large Language Models: an evaluation study","date":"2024-01-19","arxiv_id":"2401.11052","n_code_links":1,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-question-answering","title":"Reinforcement learning for question answering in programming domain using public community scoring as a human feedback","date":"2024-01-19","arxiv_id":"2401.10882","n_code_links":0,"syntology":null},{"paper":"/paper/chatqa-building-gpt-4-level-conversational-qa","slug":"chatqa-building-gpt-4-level-conversational-qa","title":"ChatQA: Surpassing GPT-4 on Conversational QA and RAG","date":"2024-01-18","arxiv_id":"2401.10225","n_code_links":0,"syntology":null},{"paper":"/paper/code-prompting-elicits-conditional-reasoning","slug":"code-prompting-elicits-conditional-reasoning","title":"Code Prompting Elicits Conditional Reasoning Abilities in Text+Code LLMs","date":"2024-01-18","arxiv_id":"2401.10065","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ukplab/arxiv2024-conditional-reasoning-llms"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gender-bias-in-machine-translation-and-the","title":"Gender Bias in Machine Translation and The Era of Large Language Models","date":"2024-01-18","arxiv_id":"2401.10016","n_code_links":0,"syntology":null},{"paper":null,"slug":"image-translation-as-diffusion-visual","title":"Image Translation as Diffusion Visual Programmers","date":"2024-01-18","arxiv_id":"2401.09742","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-biases-in-large-language-models","title":"Leveraging Biases in Large Language Models: \"bias-kNN'' for Effective Few-Shot Learning","date":"2024-01-18","arxiv_id":"2401.09783","n_code_links":0,"syntology":null},{"paper":"/paper/when-neural-code-completion-models-size-up","slug":"when-neural-code-completion-models-size-up","title":"When Neural Code Completion Models Size up the Situation: Attaining Cheaper and Faster Completion through Dynamic Model Inference","date":"2024-01-18","arxiv_id":"2401.09964","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-classification-performance-with","title":"Improving Classification Performance With Human Feedback: Label a few, we label the rest","date":"2024-01-17","arxiv_id":"2401.09555","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-emotions-demographic","slug":"learning-from-emotions-demographic","title":"Learning from Implicit User Feedback, Emotions and Demographic Information in Task-Oriented and Document-Grounded Dialogues","date":"2024-01-17","arxiv_id":"2401.09248","n_code_links":1,"syntology":null},{"paper":null,"slug":"mada-meta-adaptive-optimizers-through-hyper","title":"MADA: Meta-Adaptive Optimizers through hyper-gradient Descent","date":"2024-01-17","arxiv_id":"2401.08893","n_code_links":0,"syntology":null},{"paper":null,"slug":"application-of-llm-agents-in-recruitment-a","title":"Application of LLM Agents in Recruitment: A Novel Framework for Resume Screening","date":"2024-01-16","arxiv_id":"2401.08315","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-robustness-of-llm-synthetic-text","title":"Enhancing Robustness of LLM-Synthetic Text Detectors for Academic Writing: A Comprehensive Analysis","date":"2024-01-16","arxiv_id":"2401.08046","n_code_links":0,"syntology":null},{"paper":"/paper/exploiting-inter-layer-expert-affinity-for","slug":"exploiting-inter-layer-expert-affinity-for","title":"Exploiting Inter-Layer Expert Affinity for Accelerating Mixture-of-Experts Model Inference","date":"2024-01-16","arxiv_id":"2401.08383","n_code_links":1,"syntology":null},{"paper":null,"slug":"rag-vs-fine-tuning-pipelines-tradeoffs-and-a","title":"RAG vs Fine-tuning: Pipelines, Tradeoffs, and a Case Study on Agriculture","date":"2024-01-16","arxiv_id":"2401.08406","n_code_links":0,"syntology":null},{"paper":"/paper/rotbench-a-multi-level-benchmark-for","slug":"rotbench-a-multi-level-benchmark-for","title":"RoTBench: A Multi-Level Benchmark for Evaluating the Robustness of Large Language Models in Tool Learning","date":"2024-01-16","arxiv_id":"2401.08326","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["junjie-ye/rotbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/tuning-language-models-by-proxy","slug":"tuning-language-models-by-proxy","title":"Tuning Language Models by Proxy","date":"2024-01-16","arxiv_id":"2401.08565","n_code_links":2,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["alisawuffles/proxy-tuning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-novel-approach-for-automatic-program-repair","slug":"a-novel-approach-for-automatic-program-repair","title":"A Novel Approach for Automatic Program Repair using Round-Trip Translation with Large Language Models","date":"2024-01-15","arxiv_id":"2401.07994","n_code_links":1,"syntology":null},{"paper":"/paper/harnessing-large-language-models-over","slug":"harnessing-large-language-models-over","title":"Harnessing Large Language Models Over Transformer Models for Detecting Bengali Depressive Social Media Text: A Comprehensive Study","date":"2024-01-14","arxiv_id":"2401.07310","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-be-homo-economicus-can-an-llm","title":"Learning to be Homo Economicus: Can an LLM Learn Preferences from Choice","date":"2024-01-14","arxiv_id":"2401.07345","n_code_links":0,"syntology":null},{"paper":null,"slug":"mapgpt-map-guided-prompting-for-unified","title":"MapGPT: Map-Guided Prompting with Adaptive Path Planning for Vision-and-Language Navigation","date":"2024-01-14","arxiv_id":"2401.07314","n_code_links":0,"syntology":null},{"paper":null,"slug":"streamlining-the-selection-phase-of","title":"Streamlining the Selection Phase of Systematic Literature Reviews (SLRs) Using AI-Enabled GPT-4 Assistant API","date":"2024-01-14","arxiv_id":"2402.18582","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-multi-stage-prompting-approach-for","slug":"a-novel-multi-stage-prompting-approach-for","title":"A Novel Multi-Stage Prompting Approach for Language Agnostic MCQ Generation using GPT","date":"2024-01-13","arxiv_id":"2401.07098","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-large-language-models-in-mechanical","title":"Assessing Large Language Models in Mechanical Engineering Education: A Study on Mechanics-Focused Conceptual Understanding","date":"2024-01-13","arxiv_id":"2401.12983","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-confidence-elicitation-and-sample","title":"Combining Confidence Elicitation and Sample-based Methods for Uncertainty Quantification in Misinformation Mitigation","date":"2024-01-13","arxiv_id":"2401.08694","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-gpt-4-and-open-source-language","title":"Comparing GPT-4 and Open-Source Language Models in Misinformation Mitigation","date":"2024-01-12","arxiv_id":"2401.06920","n_code_links":0,"syntology":null},{"paper":"/paper/from-automation-to-augmentation-large","slug":"from-automation-to-augmentation-large","title":"Human-AI Collaborative Essay Scoring: A Dual-Process Framework with LLMs","date":"2024-01-12","arxiv_id":"2401.06431","n_code_links":1,"syntology":null},{"paper":"/paper/how-johnny-can-persuade-llms-to-jailbreak","slug":"how-johnny-can-persuade-llms-to-jailbreak","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety by Humanizing LLMs","date":"2024-01-12","arxiv_id":"2401.06373","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chats-lab/persuasive_jailbreaker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/intention-analysis-prompting-makes-large","slug":"intention-analysis-prompting-makes-large","title":"Intention Analysis Makes LLMs A Good Jailbreak Defender","date":"2024-01-12","arxiv_id":"2401.06561","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["alphadl/safellm_with_intentionanalysis"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/mission-impossible-language-models","slug":"mission-impossible-language-models","title":"Mission: Impossible Language Models","date":"2024-01-12","arxiv_id":"2401.06416","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jkallini/mission-impossible-language-models"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"persianmind-a-cross-lingual-persian-english","title":"PersianMind: A Cross-Lingual Persian-English Large Language Model","date":"2024-01-12","arxiv_id":"2401.06466","n_code_links":0,"syntology":null},{"paper":"/paper/pizzacommonsense-learning-to-model","slug":"pizzacommonsense-learning-to-model","title":"PizzaCommonSense: Learning to Model Commonsense Reasoning about Intermediate Steps in Cooking Recipes","date":"2024-01-12","arxiv_id":"2401.06930","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-data-contamination-for-pre","title":"Investigating Data Contamination for Pre-training Language Models","date":"2024-01-11","arxiv_id":"2401.06059","n_code_links":0,"syntology":null},{"paper":null,"slug":"mutation-based-consistency-testing-for","title":"Mutation-based Consistency Testing for Evaluating the Code Understanding Capability of LLMs","date":"2024-01-11","arxiv_id":"2401.05940","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-based-mental-health-screening-from","title":"Prompt-based mental health screening from social media text","date":"2024-01-11","arxiv_id":"2401.05912","n_code_links":0,"syntology":null},{"paper":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/autoact-automatic-agent-learning-from-scratch","slug":"autoact-automatic-agent-learning-from-scratch","title":"AutoAct: Automatic Agent Learning from Scratch for QA via Self-Planning","date":"2024-01-10","arxiv_id":"2401.05268","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zjunlp/autoact"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"8a9c44d3cab0c752dbc2bf815068cf15c301511bc04169c0bf0442c9f76187c0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}