{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/cosine-annealing/papers/16","list_of":"/method/cosine-annealing","method":"Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":16,"pages_in_order":40,"rows_per_page":100,"rows":[1501,1600],"of":3965,"counts":{"archive_papers_tagged":3965,"with_a_code_link":1734,"where_syntology_ran_a_sample":627,"not_listed_spam_title":0,"listed":3965,"listed_where_code_ran":627,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":513,"every_run_a_failure_of_syntologys_instrument":114,"listed_with_a_run_with_no_instrument_failure":513,"listed_every_run_a_failure_of_syntologys_instrument":114,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/cosine-annealing","prev":"/method/cosine-annealing/papers/15","next":"/method/cosine-annealing/papers/17","papers":[{"paper":null,"slug":"decomposed-prompting-unveiling-multilingual","title":"Decomposed Prompting: Unveiling Multilingual Linguistic Structure Knowledge in English-Centric Large Language Models","date":"2024-02-28","arxiv_id":"2402.18397","n_code_links":0,"syntology":null},{"paper":"/paper/gradient-free-adaptive-global-pruning-for-pre","slug":"gradient-free-adaptive-global-pruning-for-pre","title":"SparseLLM: Towards Global Pruning for Pre-trained Language Models","date":"2024-02-28","arxiv_id":"2402.17946","n_code_links":2,"syntology":{"ran":3,"of":9,"n_ran_checked":2,"n_instrument":1,"unverified":6,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["baithebest/adagp","baithebest/sparsellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/keeping-llms-aligned-after-fine-tuning-the","slug":"keeping-llms-aligned-after-fine-tuning-the","title":"Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates","date":"2024-02-28","arxiv_id":"2402.18540","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":5,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["vfleaking/ptst"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-language-model-based-framework-for-new","slug":"a-language-model-based-framework-for-new","title":"A Language Model based Framework for New Concept Placement in Ontologies","date":"2024-02-27","arxiv_id":"2402.17897","n_code_links":1,"syntology":null},{"paper":null,"slug":"cocoa-cbt-based-conversational-counseling","title":"COCOA: CBT-based Conversational Counseling Agent using Memory Specialized in Cognitive Distortions and Dynamic Prompt","date":"2024-02-27","arxiv_id":"2402.17546","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-learning-detection-method-for-large","title":"Deep Learning Detection Method for Large Language Models-Generated Scientific Content","date":"2024-02-27","arxiv_id":"2403.00828","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-vision-language-stem-skills-of","slug":"measuring-vision-language-stem-skills-of","title":"Measuring Vision-Language STEM Skills of Neural Models","date":"2024-02-27","arxiv_id":"2402.17205","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stemdataset/STEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/variational-learning-is-effective-for-large","slug":"variational-learning-is-effective-for-large","title":"Variational Learning is Effective for Large Deep Networks","date":"2024-02-27","arxiv_id":"2402.17641","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["team-approx-bayes/ivon"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-integrated-data-processing-framework-for","slug":"an-integrated-data-processing-framework-for","title":"An Integrated Data Processing Framework for Pretraining Foundation Models","date":"2024-02-26","arxiv_id":"2402.16358","n_code_links":2,"syntology":null},{"paper":"/paper/chatmusician-understanding-and-generating","slug":"chatmusician-understanding-and-generating","title":"ChatMusician: Understanding and Generating Music Intrinsically with LLM","date":"2024-02-25","arxiv_id":"2402.16153","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hf-lin/ChatMusician"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-learning-approaches-for-improving","title":"Deep Learning Approaches for Improving Question Answering Systems in Hepatocellular Carcinoma Research","date":"2024-02-25","arxiv_id":"2402.16038","n_code_links":0,"syntology":null},{"paper":"/paper/fusechat-knowledge-fusion-of-chat-models","slug":"fusechat-knowledge-fusion-of-chat-models","title":"Knowledge Fusion of Chat LLMs: A Preliminary Technical Report","date":"2024-02-25","arxiv_id":"2402.16107","n_code_links":2,"syntology":null},{"paper":null,"slug":"llms-can-defend-themselves-against","title":"LLMs Can Defend Themselves Against Jailbreaking in a Practical Manner: A Vision Paper","date":"2024-02-24","arxiv_id":"2402.15727","n_code_links":0,"syntology":null},{"paper":null,"slug":"look-before-you-leap-problem-elaboration","title":"Look Before You Leap: Problem Elaboration Prompting Improves Mathematical Reasoning in Large Language Models","date":"2024-02-24","arxiv_id":"2402.15764","n_code_links":0,"syntology":null},{"paper":null,"slug":"prp-propagating-universal-perturbations-to","title":"PRP: Propagating Universal Perturbations to Attack Large Language Model Guard-Rails","date":"2024-02-24","arxiv_id":"2402.15911","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-first-look-at-gpt-apps-landscape-and","title":"A First Look at GPT Apps: Landscape and Vulnerability","date":"2024-02-23","arxiv_id":"2402.15105","n_code_links":0,"syntology":null},{"paper":"/paper/advancing-parameter-efficiency-in-fine-tuning","slug":"advancing-parameter-efficiency-in-fine-tuning","title":"Advancing Parameter Efficiency in Fine-tuning via Representation Editing","date":"2024-02-23","arxiv_id":"2402.15179","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["mlwu22/red"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/attributionbench-how-hard-is-automatic","slug":"attributionbench-how-hard-is-automatic","title":"AttributionBench: How Hard is Automatic Attribution Evaluation?","date":"2024-02-23","arxiv_id":"2402.15089","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-large-language-models-for-domain","title":"Fine-tuning Large Language Models for Domain-specific Machine Translation","date":"2024-02-23","arxiv_id":"2402.15061","n_code_links":0,"syntology":null},{"paper":"/paper/prompting-llms-to-compose-meta-review-drafts","slug":"prompting-llms-to-compose-meta-review-drafts","title":"LLMs as Meta-Reviewers' Assistants: A Case Study","date":"2024-02-23","arxiv_id":"2402.15589","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-large-language-models-detect","title":"Can Large Language Models Detect Misinformation in Scientific News Reporting?","date":"2024-02-22","arxiv_id":"2402.14268","n_code_links":0,"syntology":null},{"paper":null,"slug":"copilot-evaluation-harness-evaluating-llm","title":"Copilot Evaluation Harness: Evaluating LLM-Guided Software Programming","date":"2024-02-22","arxiv_id":"2402.14261","n_code_links":0,"syntology":null},{"paper":"/paper/hint-before-solving-prompting-guiding-llms-to","slug":"hint-before-solving-prompting-guiding-llms-to","title":"Hint-before-Solving Prompting: Guiding LLMs to Effectively Utilize Encoded Knowledge","date":"2024-02-22","arxiv_id":"2402.14310","n_code_links":1,"syntology":null},{"paper":"/paper/kocosa-korean-context-aware-sarcasm-detection","slug":"kocosa-korean-context-aware-sarcasm-detection","title":"KoCoSa: Korean Context-aware Sarcasm Detection Dataset","date":"2024-02-22","arxiv_id":"2402.14428","n_code_links":1,"syntology":null},{"paper":null,"slug":"roboscript-code-generation-for-free-form","title":"RoboScript: Code Generation for Free-Form Manipulation Tasks across Real and Simulation","date":"2024-02-22","arxiv_id":"2402.14623","n_code_links":0,"syntology":null},{"paper":"/paper/tokenization-counts-the-impact-of","slug":"tokenization-counts-the-impact-of","title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","date":"2024-02-22","arxiv_id":"2402.14903","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aadityasingh/tokenizationcounts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-understanding-counseling","title":"Towards Understanding Counseling Conversations: Domain Knowledge and Large Language Models","date":"2024-02-22","arxiv_id":"2402.14200","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-evaluation-of-large-language-models-in","title":"An Evaluation of Large Language Models in Bioinformatics Research","date":"2024-02-21","arxiv_id":"2402.13714","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-hate-speech-nlp-s-challenges-and","title":"Beyond Hate Speech: NLP's Challenges and Opportunities in Uncovering Dehumanizing Language","date":"2024-02-21","arxiv_id":"2402.13818","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-efficient-transformers-really-save","title":"Do Efficient Transformers Really Save Computation?","date":"2024-02-21","arxiv_id":"2402.13934","n_code_links":0,"syntology":null},{"paper":null,"slug":"hallucinations-or-attention-misdirection-the","title":"Hallucinations or Attention Misdirection? The Path to Strategic Value Extraction in Business Using Large Language Models","date":"2024-02-21","arxiv_id":"2402.14002","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-graph-enhanced-large-language-model","title":"Knowledge Graph Enhanced Large Language Model Editing","date":"2024-02-21","arxiv_id":"2402.13593","n_code_links":0,"syntology":null},{"paper":"/paper/llm-jailbreak-attack-versus-defense","slug":"llm-jailbreak-attack-versus-defense","title":"A Comprehensive Study of Jailbreak Attack versus Defense for Large Language Models","date":"2024-02-21","arxiv_id":"2402.13457","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ltroin/llm_attack_defense_arena"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/synfac-edit-synthetic-imitation-edit-feedback","slug":"synfac-edit-synthetic-imitation-edit-feedback","title":"SYNFAC-EDIT: Synthetic Imitation Edit Feedback for Factual Alignment in Clinical Summarization","date":"2024-02-21","arxiv_id":"2402.13919","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-retrieval-augmented-generation","slug":"benchmarking-retrieval-augmented-generation","title":"Benchmarking Retrieval-Augmented Generation for Medicine","date":"2024-02-20","arxiv_id":"2402.13178","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["teddy-xionggz/medrag","teddy-xionggz/mirage"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/can-gnn-be-good-adapter-for-llms","slug":"can-gnn-be-good-adapter-for-llms","title":"Can GNN be Good Adapter for LLMs?","date":"2024-02-20","arxiv_id":"2402.12984","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zjunet/graphadapter","hxttkl/GraphAdapter"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatel-entity-linking-with-chatbots","slug":"chatel-entity-linking-with-chatbots","title":"ChatEL: Entity Linking with Chatbots","date":"2024-02-20","arxiv_id":"2402.14858","n_code_links":1,"syntology":null},{"paper":null,"slug":"evograd-a-dynamic-take-on-the-winograd-schema","title":"EvoGrad: A Dynamic Take on the Winograd Schema Challenge with Human Adversaries","date":"2024-02-20","arxiv_id":"2402.13372","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-the-system-message-really-important-to","title":"Is the System Message Really Important to Jailbreaks in Large Language Models?","date":"2024-02-20","arxiv_id":"2402.14857","n_code_links":0,"syntology":null},{"paper":"/paper/moelora-contrastive-learning-guided-mixture","slug":"moelora-contrastive-learning-guided-mixture","title":"MoELoRA: Contrastive Learning Guided Mixture of Experts on Parameter-Efficient Fine-Tuning for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12851","n_code_links":1,"syntology":null},{"paper":null,"slug":"nl2formula-generating-spreadsheet-formulas","title":"NL2Formula: Generating Spreadsheet Formulas from Natural Language Queries","date":"2024-02-20","arxiv_id":"2402.14853","n_code_links":0,"syntology":null},{"paper":"/paper/promptkd-distilling-student-friendly","slug":"promptkd-distilling-student-friendly","title":"PromptKD: Distilling Student-Friendly Knowledge for Generative Language Models via Prompt Tuning","date":"2024-02-20","arxiv_id":"2402.12842","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gmkim-ai/promptkd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/reflect-rl-two-player-online-rl-fine-tuning","slug":"reflect-rl-two-player-online-rl-fine-tuning","title":"Reflect-RL: Two-Player Online RL Fine-Tuning for LMs","date":"2024-02-20","arxiv_id":"2402.12621","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":1,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zhourunlong/reflect-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sql-craft-text-to-sql-through-interactive","title":"$R^3$: \"This is My SQL, Are You With Me?\" A Consensus-Based Multi-Agent System for Text-to-SQL Tasks","date":"2024-02-20","arxiv_id":"2402.14851","n_code_links":0,"syntology":null},{"paper":"/paper/the-impact-of-demonstrations-on-multilingual","slug":"the-impact-of-demonstrations-on-multilingual","title":"The Impact of Demonstrations on Multilingual In-Context Learning: A Multidimensional Analysis","date":"2024-02-20","arxiv_id":"2402.12976","n_code_links":1,"syntology":null},{"paper":"/paper/a-critical-evaluation-of-ai-feedback-for","slug":"a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["architsharma97/dpo-rlaif"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/analobench-benchmarking-the-identification-of","slug":"analobench-benchmarking-the-identification-of","title":"AnaloBench: Benchmarking the Identification of Abstract and Long-context Analogies","date":"2024-02-19","arxiv_id":"2402.12370","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jhu-clsp/analogical-reasoning","JHU-CLSP/AnaloBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ask-optimal-questions-aligning-large-language","title":"Ask Optimal Questions: Aligning Large Language Models with Retriever's Preference in Conversational Search","date":"2024-02-19","arxiv_id":"2402.11827","n_code_links":0,"syntology":null},{"paper":null,"slug":"deepcode-ai-fix-fixing-security","title":"DeepCode AI Fix: Fixing Security Vulnerabilities with Large Language Models","date":"2024-02-19","arxiv_id":"2402.13291","n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-multilingual-fact-checking-at","title":"Surprising Efficacy of Fine-Tuned Transformers for Fact-Checking over Larger Language Models","date":"2024-02-19","arxiv_id":"2402.12147","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-open-source-there-yet-a-comparative-study","title":"Is Open-Source There Yet? A Comparative Study on Commercial and Open-Source LLMs in Their Ability to Label Chest X-Ray Reports","date":"2024-02-19","arxiv_id":"2402.12298","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-ranking-less-capable-language-models-are","title":"Enabling Weak LLMs to Judge Response Reliability via Meta Ranking","date":"2024-02-19","arxiv_id":"2402.12146","n_code_links":0,"syntology":null},{"paper":"/paper/query-based-adversarial-prompt-generation","slug":"query-based-adversarial-prompt-generation","title":"Query-Based Adversarial Prompt Generation","date":"2024-02-19","arxiv_id":"2402.12329","n_code_links":2,"syntology":null},{"paper":null,"slug":"spml-a-dsl-for-defending-language-models","title":"SPML: A DSL for Defending Language Models Against Prompt Attacks","date":"2024-02-19","arxiv_id":"2402.11755","n_code_links":0,"syntology":null},{"paper":null,"slug":"stick-to-your-role-stability-of-personal","title":"Stick to your Role! Stability of Personal Values Expressed in Large Language Models","date":"2024-02-19","arxiv_id":"2402.14846","n_code_links":0,"syntology":null},{"paper":null,"slug":"your-large-language-model-is-secretly-a","title":"Your Large Language Model is Secretly a Fairness Proponent and You Should Prompt it Like One","date":"2024-02-19","arxiv_id":"2402.12150","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-news-narratives-a-critical-analysis","title":"Decoding News Narratives: A Critical Analysis of Large Language Models in Framing Detection","date":"2024-02-18","arxiv_id":"2402.11621","n_code_links":0,"syntology":null},{"paper":"/paper/gnnavi-navigating-the-information-flow-in","slug":"gnnavi-navigating-the-information-flow-in","title":"GNNavi: Navigating the Information Flow in Large Language Models by Graph Neural Network","date":"2024-02-18","arxiv_id":"2402.11709","n_code_links":1,"syntology":null},{"paper":"/paper/perils-of-self-feedback-self-bias-amplifies","slug":"perils-of-self-feedback-self-bias-amplifies","title":"Pride and Prejudice: LLM Amplifies Self-Bias in Self-Refinement","date":"2024-02-18","arxiv_id":"2402.11436","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xu1998hz/llm_self_bias"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-chatgpt-for-next-generation","title":"Exploring ChatGPT for Next-generation Information Retrieval: Opportunities and Challenges","date":"2024-02-17","arxiv_id":"2402.11203","n_code_links":0,"syntology":null},{"paper":null,"slug":"gendec-a-robust-generative-question","title":"GenDec: A robust generative Question-decomposition method for Multi-hop reasoning","date":"2024-02-17","arxiv_id":"2402.11166","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-reasoning-abilities-of-chatgpt","title":"Assessing the Reasoning Abilities of ChatGPT in the Context of Claim Verification","date":"2024-02-16","arxiv_id":"2402.10735","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-separators-improve-chain-of-thought","title":"Can Separators Improve Chain-of-Thought Prompting?","date":"2024-02-16","arxiv_id":"2402.10645","n_code_links":0,"syntology":null},{"paper":"/paper/disordered-dabs-a-benchmark-for-dynamic","slug":"disordered-dabs-a-benchmark-for-dynamic","title":"Disordered-DABS: A Benchmark for Dynamic Aspect-Based Summarization in Disordered Texts","date":"2024-02-16","arxiv_id":"2402.10554","n_code_links":1,"syntology":null},{"paper":"/paper/in-search-of-needles-in-a-10m-haystack","slug":"in-search-of-needles-in-a-10m-haystack","title":"In Search of Needles in a 11M Haystack: Recurrent Memory Finds What LLMs Miss","date":"2024-02-16","arxiv_id":"2402.10790","n_code_links":2,"syntology":null},{"paper":null,"slug":"inference-to-the-best-explanation-in-large","title":"Inference to the Best Explanation in Large Language Models","date":"2024-02-16","arxiv_id":"2402.10767","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-as-zero-shot-dialogue","slug":"large-language-models-as-zero-shot-dialogue","title":"Large Language Models as Zero-shot Dialogue State Tracker through Function Calling","date":"2024-02-16","arxiv_id":"2402.10466","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["facebookresearch/fnctod"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"large-language-models-fall-short","title":"Large Language Models Fall Short: Understanding Complex Relationships in Detective Narratives","date":"2024-02-16","arxiv_id":"2402.11051","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-in-the-heart-of-differential-testing-a","title":"LLMs in the Heart of Differential Testing: A Case Study on a Medical Rule Engine","date":"2024-02-16","arxiv_id":"2404.03664","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-cultural-commonsense-knowledge","title":"Cultural Commonsense Knowledge for Intercultural Dialogues","date":"2024-02-16","arxiv_id":"2402.10689","n_code_links":0,"syntology":null},{"paper":"/paper/network-formation-and-dynamics-among-multi","slug":"network-formation-and-dynamics-among-multi","title":"Network Formation and Dynamics Among Multi-LLMs","date":"2024-02-16","arxiv_id":"2402.10659","n_code_links":1,"syntology":null},{"paper":"/paper/universal-prompt-optimizer-for-safe-text-to","slug":"universal-prompt-optimizer-for-safe-text-to","title":"Universal Prompt Optimizer for Safe Text-to-Image Generation","date":"2024-02-16","arxiv_id":"2402.10882","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-analysis-of-langauge-frequency-and-error","title":"An Analysis of Language Frequency and Error Correction for Esperanto","date":"2024-02-15","arxiv_id":"2402.09696","n_code_links":0,"syntology":null},{"paper":null,"slug":"best-arm-identification-for-prompt-learning","title":"Efficient Prompt Optimization Through the Lens of Best Arm Identification","date":"2024-02-15","arxiv_id":"2402.09723","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-large-language-model-llm","title":"Fine-tuning Large Language Model (LLM) Artificial Intelligence Chatbots in Ophthalmology and LLM-based evaluation using GPT-4","date":"2024-02-15","arxiv_id":"2402.10083","n_code_links":0,"syntology":null},{"paper":"/paper/pal-proxy-guided-black-box-attack-on-large","slug":"pal-proxy-guided-black-box-attack-on-large","title":"PAL: Proxy-Guided Black-Box Attack on Large Language Models","date":"2024-02-15","arxiv_id":"2402.09674","n_code_links":1,"syntology":null},{"paper":"/paper/the-butterfly-effect-of-model-editing-few","slug":"the-butterfly-effect-of-model-editing-few","title":"The Butterfly Effect of Model Editing: Few Edits Can Trigger Large Language Models Collapse","date":"2024-02-15","arxiv_id":"2402.09656","n_code_links":1,"syntology":null},{"paper":"/paper/api-pack-a-massive-multilingual-dataset-for","slug":"api-pack-a-massive-multilingual-dataset-for","title":"API Pack: A Massive Multi-Programming Language Dataset for API Call Generation","date":"2024-02-14","arxiv_id":"2402.09615","n_code_links":1,"syntology":null},{"paper":"/paper/mpirigen-mpi-code-generation-through-domain","slug":"mpirigen-mpi-code-generation-through-domain","title":"MPIrigen: MPI Code Generation through Domain-Specific Language Models","date":"2024-02-14","arxiv_id":"2402.09126","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-counterfactual-tasks-to-evaluate-the","title":"Using Counterfactual Tasks to Evaluate the Generality of Analogical Reasoning in Large Language Models","date":"2024-02-14","arxiv_id":"2402.08955","n_code_links":0,"syntology":null},{"paper":null,"slug":"auditing-counterfire-evaluating-advanced","title":"\"Reasoning\" with Rhetoric: On the Style-Evidence Tradeoff in LLM-Generated Counter-Arguments","date":"2024-02-13","arxiv_id":"2402.08498","n_code_links":0,"syntology":null},{"paper":"/paper/cold-attack-jailbreaking-llms-with","slug":"cold-attack-jailbreaking-llms-with","title":"COLD-Attack: Jailbreaking LLMs with Stealthiness and Controllability","date":"2024-02-13","arxiv_id":"2402.08679","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yu-fangxu/cold-attack"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"eliciting-big-five-personality-traits-in","title":"Eliciting Personality Traits in Large Language Models","date":"2024-02-13","arxiv_id":"2402.08341","n_code_links":0,"syntology":null},{"paper":null,"slug":"lying-blindly-bypassing-chatgpt-s-safeguards","title":"Lying Blindly: Bypassing ChatGPT's Safeguards to Generate Hard-to-Detect Disinformation Claims","date":"2024-02-13","arxiv_id":"2402.08467","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-and-controlling-instruction-in","slug":"measuring-and-controlling-instruction-in","title":"Measuring and Controlling Instruction (In)Stability in Language Model Dialogs","date":"2024-02-13","arxiv_id":"2402.10962","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["likenneth/persona_drift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mitigating-object-hallucination-in-large","title":"Mitigating Object Hallucination in Large Vision-Language Models via Classifier-Free Guidance","date":"2024-02-13","arxiv_id":"2402.08680","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-optimization-in-multi-step-tasks","slug":"prompt-optimization-in-multi-step-tasks","title":"PRompt Optimization in Multi-Step Tasks (PROMST): Integrating Human Feedback and Heuristic-based Sampling","date":"2024-02-13","arxiv_id":"2402.08702","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yongchao98/promst"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/addressing-cognitive-bias-in-medical-language","slug":"addressing-cognitive-bias-in-medical-language","title":"Addressing cognitive bias in medical language models","date":"2024-02-12","arxiv_id":"2402.08113","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["carlwharris/cog-bias-med-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/breakgpt-a-large-language-model-with-multi","slug":"breakgpt-a-large-language-model-with-multi","title":"BreakGPT: A Large Language Model with Multi-stage Structure for Financial Breakout Detection","date":"2024-02-12","arxiv_id":"2402.07536","n_code_links":1,"syntology":null},{"paper":"/paper/cybermetric-a-benchmark-dataset-for","slug":"cybermetric-a-benchmark-dataset-for","title":"CyberMetric: A Benchmark Dataset based on Retrieval-Augmented Generation for Evaluating LLMs in Cybersecurity Knowledge","date":"2024-02-12","arxiv_id":"2402.07688","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-the-impact-of-data","title":"Investigating the Impact of Data Contamination of Large Language Models in Text-to-SQL Translation","date":"2024-02-12","arxiv_id":"2402.08100","n_code_links":0,"syntology":null},{"paper":null,"slug":"sequential-ordering-in-textual-descriptions","title":"Can Graph Descriptive Order Affect Solving Graph Problems with LLMs?","date":"2024-02-11","arxiv_id":"2402.07140","n_code_links":0,"syntology":null},{"paper":"/paper/chemllm-a-chemical-large-language-model","slug":"chemllm-a-chemical-large-language-model","title":"ChemLLM: A Chemical Large Language Model","date":"2024-02-10","arxiv_id":"2402.06852","n_code_links":1,"syntology":null},{"paper":null,"slug":"sentinels-of-the-stream-unleashing-large","title":"Sentinels of the Stream: Unleashing Large Language Models for Dynamic Packet Classification in Software Defined Networks -- Position Paper","date":"2024-02-10","arxiv_id":"2402.07950","n_code_links":0,"syntology":null},{"paper":"/paper/culturellm-incorporating-cultural-differences","slug":"culturellm-incorporating-cultural-differences","title":"CultureLLM: Incorporating Cultural Differences into Large Language Models","date":"2024-02-09","arxiv_id":"2402.10946","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["scarelette/culturellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/entgpt-linking-generative-large-language","slug":"entgpt-linking-generative-large-language","title":"EntGPT: Linking Generative Large Language Models with Knowledge Bases","date":"2024-02-09","arxiv_id":"2402.06738","n_code_links":1,"syntology":null},{"paper":"/paper/exaranker-open-synthetic-explanation-for-ir","slug":"exaranker-open-synthetic-explanation-for-ir","title":"ExaRanker-Open: Synthetic Explanation for IR using Open-Source LLMs","date":"2024-02-09","arxiv_id":"2402.06334","n_code_links":1,"syntology":null},{"paper":null,"slug":"learn-to-be-efficient-build-structured","title":"Learn To be Efficient: Build Structured Sparsity in Large Language Models","date":"2024-02-09","arxiv_id":"2402.06126","n_code_links":0,"syntology":null},{"paper":"/paper/comprehensive-assessment-of-jailbreak-attacks","slug":"comprehensive-assessment-of-jailbreak-attacks","title":"JailbreakRadar: Comprehensive Assessment of Jailbreak Attacks Against LLMs","date":"2024-02-08","arxiv_id":"2402.05668","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["TrustAIRLab/Comprehensive_Jailbreak_Assessment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/in-context-principle-learning-from-mistakes","slug":"in-context-principle-learning-from-mistakes","title":"In-Context Principle Learning from Mistakes","date":"2024-02-08","arxiv_id":"2402.05403","n_code_links":1,"syntology":null}],"record_sha256":"33de1c387b02f8edf706b3bf72a0d94a9425c852bdaf66d21462e9eb846ab10e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}