{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/llama/papers/3","list_of":"/method/llama","method":"LLaMA","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":3,"pages_in_order":11,"rows_per_page":100,"rows":[201,300],"of":1062,"counts":{"archive_papers_tagged":1062,"with_a_code_link":423,"where_syntology_ran_a_sample":143,"not_listed_spam_title":0,"listed":1062,"listed_where_code_ran":143,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":127,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":127,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/llama","prev":"/method/llama/papers/2","next":"/method/llama/papers/4","papers":[{"paper":"/paper/llm-benchmarking-with-llama2-evaluating-code","slug":"llm-benchmarking-with-llama2-evaluating-code","title":"LLM Benchmarking with LLaMA2: Evaluating Code Development Performance Across Multiple Programming Languages","date":"2025-03-24","arxiv_id":"2503.19217","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-recent-large-language-models","title":"Investigating Recent Large Language Models for Vietnamese Machine Reading Comprehension","date":"2025-03-23","arxiv_id":"2503.18062","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-specific-neurons-do-not-facilitate","title":"Language-specific Neurons Do Not Facilitate Cross-Lingual Transfer","date":"2025-03-21","arxiv_id":"2503.17456","n_code_links":0,"syntology":null},{"paper":null,"slug":"saudiculture-a-benchmark-for-evaluating-large","title":"SaudiCulture: A Benchmark for Evaluating Large Language Models Cultural Competence within Saudi Arabia","date":"2025-03-21","arxiv_id":"2503.17485","n_code_links":0,"syntology":null},{"paper":null,"slug":"text2model-generating-dynamic-chemical","title":"Text2Model: Generating dynamic chemical reactor models using large language models (LLMs)","date":"2025-03-21","arxiv_id":"2503.17004","n_code_links":0,"syntology":null},{"paper":null,"slug":"v-seek-accelerating-llm-reasoning-on-open","title":"V-Seek: Accelerating LLM Reasoning on Open-hardware Server-class RISC-V Platforms","date":"2025-03-21","arxiv_id":"2503.17422","n_code_links":0,"syntology":null},{"paper":"/paper/variance-control-via-weight-rescaling-in-llm","slug":"variance-control-via-weight-rescaling-in-llm","title":"Variance Control via Weight Rescaling in LLM Pre-training","date":"2025-03-21","arxiv_id":"2503.17500","n_code_links":1,"syntology":null},{"paper":null,"slug":"only-a-little-to-the-left-a-theory-grounded","title":"Only a Little to the Left: A Theory-grounded Measure of Political Bias in Large Language Models","date":"2025-03-20","arxiv_id":"2503.16148","n_code_links":0,"syntology":null},{"paper":"/paper/poly-fever-a-multilingual-fact-verification","slug":"poly-fever-a-multilingual-fact-verification","title":"Poly-FEVER: A Multilingual Fact Verification Benchmark for Hallucination Detection in Large Language Models","date":"2025-03-19","arxiv_id":"2503.16541","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-environment-with-llm","title":"Reinforcement Learning Environment with LLM-Controlled Adversary in D&D 5th Edition Combat","date":"2025-03-19","arxiv_id":"2503.15726","n_code_links":0,"syntology":null},{"paper":"/paper/empowering-smaller-models-tuning-llama-and","slug":"empowering-smaller-models-tuning-llama-and","title":"Empowering Smaller Models: Tuning LLaMA and Gemma with Chain-of-Thought for Ukrainian Exam Tasks","date":"2025-03-18","arxiv_id":"2503.13988","n_code_links":1,"syntology":null},{"paper":null,"slug":"roma-a-read-only-memory-based-accelerator-for","title":"ROMA: a Read-Only-Memory-based Accelerator for QLoRA-based On-Device LLM","date":"2025-03-17","arxiv_id":"2503.12988","n_code_links":0,"syntology":null},{"paper":null,"slug":"verileaky-navigating-ip-protection-vs-utility","title":"VeriLeaky: Navigating IP Protection vs Utility in Fine-Tuning for LLM-Driven Verilog Coding","date":"2025-03-17","arxiv_id":"2503.13116","n_code_links":0,"syntology":null},{"paper":"/paper/llm-driven-multi-step-translation-from-c-to","slug":"llm-driven-multi-step-translation-from-c-to","title":"LLM-Driven Multi-step Translation from C to Rust using Static Analysis","date":"2025-03-16","arxiv_id":"2503.12511","n_code_links":1,"syntology":null},{"paper":null,"slug":"integrating-chain-of-thought-and-retrieval","title":"Integrating Chain-of-Thought and Retrieval Augmented Generation Enhances Rare Disease Diagnosis from Clinical Notes","date":"2025-03-15","arxiv_id":"2503.12286","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-large-language-models-for","title":"Optimizing Large Language Models for Detecting Symptoms of Comorbid Depression or Anxiety in Chronic Diseases: Insights from Patient Messages","date":"2025-03-14","arxiv_id":"2503.11384","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-sentiment-the-catalyst-for-llm-change","title":"Prompt Sentiment: The Catalyst for LLM Change","date":"2025-03-14","arxiv_id":"2503.13510","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-extreme-pruning-of-llms-with-plug-and","title":"Towards Extreme Pruning of LLMs with Plug-and-Play Mixed Sparsity","date":"2025-03-14","arxiv_id":"2503.11164","n_code_links":0,"syntology":null},{"paper":"/paper/scalable-evaluation-of-online-moderation","slug":"scalable-evaluation-of-online-moderation","title":"Scalable Evaluation of Online Facilitation Strategies via Synthetic Simulation of Discussions","date":"2025-03-13","arxiv_id":"2503.16505","n_code_links":1,"syntology":null},{"paper":"/paper/aligning-to-what-limits-to-rlhf-based","slug":"aligning-to-what-limits-to-rlhf-based","title":"Aligning to What? Limits to RLHF Based Alignment","date":"2025-03-12","arxiv_id":"2503.09025","n_code_links":1,"syntology":null},{"paper":null,"slug":"battling-misinformation-an-empirical-study-on","title":"Battling Misinformation: An Empirical Study on Adversarial Factuality in Open-Source Large Language Models","date":"2025-03-12","arxiv_id":"2503.10690","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-a-society-of-generative-agents-simulate","title":"Can A Society of Generative Agents Simulate Human Behavior and Inform Public Health Policy? A Case Study on Vaccine Hesitancy","date":"2025-03-12","arxiv_id":"2503.09639","n_code_links":0,"syntology":null},{"paper":"/paper/cyberllminstruct-a-new-dataset-for-analysing","slug":"cyberllminstruct-a-new-dataset-for-analysing","title":"CyberLLMInstruct: A New Dataset for Analysing Safety of Fine-Tuned LLMs Using Cyber Security Data","date":"2025-03-12","arxiv_id":"2503.09334","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-high-quality-code-generation-in","slug":"enhancing-high-quality-code-generation-in","title":"Enhancing High-Quality Code Generation in Large Language Models with Comparative Prefix-Tuning","date":"2025-03-12","arxiv_id":"2503.09020","n_code_links":1,"syntology":null},{"paper":"/paper/improving-the-reusability-of-conversational","slug":"improving-the-reusability-of-conversational","title":"Improving the Reusability of Conversational Search Test Collections","date":"2025-03-12","arxiv_id":"2503.09899","n_code_links":1,"syntology":null},{"paper":null,"slug":"surgicalvlm-agent-towards-an-interactive-ai","title":"SurgicalVLM-Agent: Towards an Interactive AI Co-Pilot for Pituitary Surgery","date":"2025-03-12","arxiv_id":"2503.09474","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-large-language-models-for-hardware","slug":"enhancing-large-language-models-for-hardware","title":"Enhancing Large Language Models for Hardware Verification: A Novel SystemVerilog Assertion Dataset","date":"2025-03-11","arxiv_id":"2503.08923","n_code_links":1,"syntology":null},{"paper":null,"slug":"fact-checking-with-generative-ai-a-systematic","title":"Fact-checking with Generative AI: A Systematic Cross-Topic Examination of LLMs Capacity to Detect Veracity of Political Information","date":"2025-03-11","arxiv_id":"2503.08404","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-virtual-users-and-bias-predicting-any","title":"Llms, Virtual Users, and Bias: Predicting Any Survey Question Without Human Data","date":"2025-03-11","arxiv_id":"2503.16498","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-llama-3-2-for-software","title":"Evaluating LLaMA 3.2 for Software Vulnerability Detection","date":"2025-03-10","arxiv_id":"2503.07770","n_code_links":0,"syntology":null},{"paper":null,"slug":"fully-autonomous-programming-using-iterative","title":"Fully Autonomous Programming using Iterative Multi-Agent Debugging with Large Language Models","date":"2025-03-10","arxiv_id":"2503.07693","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-non-replicable-social-science","title":"Identifying Non-Replicable Social Science Studies with Language Models","date":"2025-03-10","arxiv_id":"2503.10671","n_code_links":0,"syntology":null},{"paper":"/paper/roamify-designing-and-evaluating-an-llm-based","slug":"roamify-designing-and-evaluating-an-llm-based","title":"Roamify: Designing and Evaluating an LLM Based Google Chrome Extension for Personalised Itinerary Planning","date":"2025-03-10","arxiv_id":"2504.10489","n_code_links":1,"syntology":null},{"paper":"/paper/sometimes-the-model-doth-preach-quantifying","slug":"sometimes-the-model-doth-preach-quantifying","title":"Sometimes the Model doth Preach: Quantifying Religious Bias in Open LLMs through Demographic Analysis in Asian Nations","date":"2025-03-10","arxiv_id":"2503.07510","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-superior-quantization-accuracy-a","title":"Towards Superior Quantization Accuracy: A Layer-sensitive Approach","date":"2025-03-09","arxiv_id":"2503.06518","n_code_links":0,"syntology":null},{"paper":"/paper/training-llm-based-tutors-to-improve-student","slug":"training-llm-based-tutors-to-improve-student","title":"Training LLM-based Tutors to Improve Student Learning Outcomes in Dialogues","date":"2025-03-09","arxiv_id":"2503.06424","n_code_links":1,"syntology":null},{"paper":null,"slug":"critical-foreign-policy-decisions-cfpd","title":"Critical Foreign Policy Decisions (CFPD)-Benchmark: Measuring Diplomatic Preferences in Large Language Models","date":"2025-03-08","arxiv_id":"2503.06263","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-diffuser-for-red-teaming-large","title":"Reinforced Diffuser for Red Teaming Large Vision-Language Models","date":"2025-03-08","arxiv_id":"2503.06223","n_code_links":0,"syntology":null},{"paper":"/paper/this-is-your-doge-if-it-please-you-exploring","slug":"this-is-your-doge-if-it-please-you-exploring","title":"This Is Your Doge, If It Please You: Exploring Deception and Robustness in Mixture of LLMs","date":"2025-03-07","arxiv_id":"2503.05856","n_code_links":1,"syntology":null},{"paper":null,"slug":"helpsteer3-human-annotated-feedback-and-edit","title":"HelpSteer3: Human-Annotated Feedback and Edit Data to Empower Inference-Time Scaling in Open-Ended General-Domain Tasks","date":"2025-03-06","arxiv_id":"2503.04378","n_code_links":0,"syntology":null},{"paper":null,"slug":"memory-is-all-you-need-testing-how-model","title":"Memory Is All You Need: Testing How Model Memory Affects LLM Performance in Annotation Tasks","date":"2025-03-06","arxiv_id":"2503.04874","n_code_links":0,"syntology":null},{"paper":null,"slug":"pokechamp-an-expert-level-minimax-language","title":"PokéChamp: an Expert-level Minimax Language Agent","date":"2025-03-06","arxiv_id":"2503.04094","n_code_links":0,"syntology":null},{"paper":"/paper/wanda-pruning-large-language-models-via-1","slug":"wanda-pruning-large-language-models-via-1","title":"Wanda++: Pruning Large Language Models via Regional Gradients","date":"2025-03-06","arxiv_id":"2503.04992","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-in-finance-estimating","title":"Large language models in finance : what is financial sentiment?","date":"2025-03-05","arxiv_id":"2503.03612","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-multi-label-classification-of","title":"Zero-Shot Multi-Label Classification of Bangla Documents: Large Decoders Vs. Classic Encoders","date":"2025-03-04","arxiv_id":"2503.02993","n_code_links":0,"syntology":null},{"paper":null,"slug":"2503-01814","title":"LLMInit: A Free Lunch from Large Language Models for Selective Initialization of Recommendation","date":"2025-03-03","arxiv_id":"2503.01814","n_code_links":0,"syntology":null},{"paper":"/paper/cancer-type-stage-and-prognosis-assessment","slug":"cancer-type-stage-and-prognosis-assessment","title":"Cancer Type, Stage and Prognosis Assessment from Pathology Reports using LLMs","date":"2025-03-03","arxiv_id":"2503.01194","n_code_links":1,"syntology":null},{"paper":"/paper/cognitive-behaviors-that-enable-self","slug":"cognitive-behaviors-that-enable-self","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","date":"2025-03-03","arxiv_id":"2503.01307","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kanishkg/cognitive-behaviors"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"2503-00992","title":"Evidence of conceptual mastery in the application of rules by Large Language Models","date":"2025-03-02","arxiv_id":"2503.00992","n_code_links":0,"syntology":null},{"paper":null,"slug":"ladder-self-improving-llms-through-recursive","title":"LADDER: Self-Improving LLMs Through Recursive Problem Decomposition","date":"2025-03-02","arxiv_id":"2503.00735","n_code_links":0,"syntology":null},{"paper":null,"slug":"collective-reasoning-among-llms-a-framework","title":"Collective Reasoning Among LLMs A Framework for Answer Validation Without Ground Truth","date":"2025-02-28","arxiv_id":"2502.20758","n_code_links":0,"syntology":null},{"paper":"/paper/nutrigen-personalized-meal-plan-generator","slug":"nutrigen-personalized-meal-plan-generator","title":"NutriGen: Personalized Meal Plan Generator Leveraging Large Language Models to Enhance Dietary and Nutritional Adherence","date":"2025-02-28","arxiv_id":"2502.20601","n_code_links":1,"syntology":null},{"paper":"/paper/optimizing-large-language-models-for-esg","slug":"optimizing-large-language-models-for-esg","title":"Optimizing Large Language Models for ESG Activity Detection in Financial Texts","date":"2025-02-28","arxiv_id":"2502.21112","n_code_links":1,"syntology":null},{"paper":null,"slug":"passage-query-methods-for-retrieval-and","title":"Passage Query Methods for Retrieval and Reranking in Conversational Agents","date":"2025-02-28","arxiv_id":"2503.00238","n_code_links":0,"syntology":null},{"paper":"/paper/ruccod-towards-automated-icd-coding-in","slug":"ruccod-towards-automated-icd-coding-in","title":"RuCCoD: Towards Automated ICD Coding in Russian","date":"2025-02-28","arxiv_id":"2502.21263","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-exploration-of-features-to-improve-the","title":"An exploration of features to improve the generalisability of fake news detection models","date":"2025-02-27","arxiv_id":"2502.20299","n_code_links":0,"syntology":null},{"paper":null,"slug":"halora-hardware-aware-low-rank-adaptation-for","title":"HaLoRA: Hardware-aware Low-Rank Adaptation for Large Language Models Based on Hybrid Compute-in-Memory Architecture","date":"2025-02-27","arxiv_id":"2502.19747","n_code_links":0,"syntology":null},{"paper":"/paper/skippipe-partial-and-reordered-pipelining","slug":"skippipe-partial-and-reordered-pipelining","title":"SkipPipe: Partial and Reordered Pipelining Framework for Training LLMs in Heterogeneous Networks","date":"2025-02-27","arxiv_id":"2502.19913","n_code_links":1,"syntology":null},{"paper":null,"slug":"deep-bench-deep-learning-benchmark-dataset","title":"Deep-Bench: Deep Learning Benchmark Dataset for Code Generation","date":"2025-02-26","arxiv_id":"2502.18726","n_code_links":0,"syntology":null},{"paper":"/paper/neobert-a-next-generation-bert","slug":"neobert-a-next-generation-bert","title":"NeoBERT: A Next-Generation BERT","date":"2025-02-26","arxiv_id":"2502.19587","n_code_links":1,"syntology":null},{"paper":null,"slug":"nexus-o-an-omni-perceptive-and-interactive","title":"Nexus: An Omni-Perceptive And -Interactive Model for Language, Audio, And Vision","date":"2025-02-26","arxiv_id":"2503.01879","n_code_links":0,"syntology":null},{"paper":"/paper/starjob-dataset-for-llm-driven-job-shop","slug":"starjob-dataset-for-llm-driven-job-shop","title":"Starjob: Dataset for LLM-Driven Job Shop Scheduling","date":"2025-02-26","arxiv_id":"2503.01877","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-sharpness-disparity-principle-in","title":"The Sharpness Disparity Principle in Transformers for Accelerating Language Model Pre-Training","date":"2025-02-26","arxiv_id":"2502.19002","n_code_links":0,"syntology":null},{"paper":null,"slug":"ampo-active-multi-preference-optimization","title":"AMPO: Active Multi-Preference Optimization","date":"2025-02-25","arxiv_id":"2502.18293","n_code_links":0,"syntology":null},{"paper":null,"slug":"frida-to-the-rescue-analyzing-synthetic-data","title":"FRIDA to the Rescue! Analyzing Synthetic Data Effectiveness in Object-Based Common Sense Reasoning for Disaster Response","date":"2025-02-25","arxiv_id":"2502.18452","n_code_links":0,"syntology":null},{"paper":null,"slug":"nusaaksara-a-multimodal-and-multilingual","title":"NusaAksara: A Multimodal and Multilingual Benchmark for Preserving Indonesian Indigenous Scripts","date":"2025-02-25","arxiv_id":"2502.18148","n_code_links":0,"syntology":null},{"paper":null,"slug":"single-vs-dual-prompt-dialogue-generation","title":"Single- vs. Dual-Prompt Dialogue Generation with LLMs for Job Interviews in Human Resources","date":"2025-02-25","arxiv_id":"2502.18650","n_code_links":0,"syntology":null},{"paper":null,"slug":"swe-rl-advancing-llm-reasoning-via","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","date":"2025-02-25","arxiv_id":"2502.18449","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-large-language-models-good-data","title":"Are Large Language Models Good Data Preprocessors?","date":"2025-02-24","arxiv_id":"2502.16790","n_code_links":0,"syntology":null},{"paper":null,"slug":"correlating-and-predicting-human-evaluations","title":"Correlating and Predicting Human Evaluations of Language Models from Natural Language Processing Benchmarks","date":"2025-02-24","arxiv_id":"2502.18339","n_code_links":0,"syntology":null},{"paper":"/paper/cot-uq-improving-response-wise-uncertainty","slug":"cot-uq-improving-response-wise-uncertainty","title":"CoT-UQ: Improving Response-wise Uncertainty Quantification in LLMs with Chain-of-Thought","date":"2025-02-24","arxiv_id":"2502.17214","n_code_links":1,"syntology":null},{"paper":"/paper/statllm-a-dataset-for-evaluating-the","slug":"statllm-a-dataset-for-evaluating-the","title":"StatLLM: A Dataset for Evaluating the Performance of Large Language Models in Statistical Analysis","date":"2025-02-24","arxiv_id":"2502.17657","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-llm-routing-and-selection-based-on","title":"Dynamic LLM Routing and Selection based on User Preferences: Balancing Performance, Cost, and Ethics","date":"2025-02-23","arxiv_id":"2502.16696","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-multi-agent-framework-for-automated","title":"A Multi-Agent Framework for Automated Vulnerability Detection and Repair in Solidity and Move Smart Contracts","date":"2025-02-22","arxiv_id":"2502.18515","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-qwen-2-5-3b-for-realistic-movie","title":"Fine-Tuning Qwen 2.5 3B for Realistic Movie Dialogue Generation","date":"2025-02-22","arxiv_id":"2502.16274","n_code_links":0,"syntology":null},{"paper":null,"slug":"toward-a-flexible-framework-for-linear","title":"Toward a Flexible Framework for Linear Representation Hypothesis Using Maximum Likelihood Estimation","date":"2025-02-22","arxiv_id":"2502.16385","n_code_links":0,"syntology":null},{"paper":null,"slug":"automedprompt-a-new-framework-for-optimizing","title":"AutoMedPrompt: A New Framework for Optimizing LLM Medical Prompts Using Textual Gradients","date":"2025-02-21","arxiv_id":"2502.15944","n_code_links":0,"syntology":null},{"paper":null,"slug":"comprehensive-analysis-of-transparency-and","title":"Comprehensive Analysis of Transparency and Accessibility of ChatGPT, DeepSeek, And other SoTA Large Language Models","date":"2025-02-21","arxiv_id":"2502.18505","n_code_links":0,"syntology":null},{"paper":null,"slug":"mmrag-multi-mode-retrieval-augmented","title":"MMRAG: Multi-Mode Retrieval-Augmented Generation with Large Language Models for Biomedical In-Context Learning","date":"2025-02-21","arxiv_id":"2502.15954","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-llms-consider-security-an-empirical-study","title":"Do LLMs Consider Security? An Empirical Study on Responses to Programming Questions","date":"2025-02-20","arxiv_id":"2502.14202","n_code_links":0,"syntology":null},{"paper":"/paper/english-please-evaluating-machine-translation","slug":"english-please-evaluating-machine-translation","title":"English Please: Evaluating Machine Translation with Large Language Models for Multilingual Bug Reports","date":"2025-02-20","arxiv_id":"2502.14338","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-conversational-agents-with-theory","slug":"enhancing-conversational-agents-with-theory","title":"Enhancing Conversational Agents with Theory of Mind: Aligning Beliefs, Desires, and Intentions for Human-Like Interaction","date":"2025-02-20","arxiv_id":"2502.14171","n_code_links":1,"syntology":null},{"paper":null,"slug":"less-is-more-improving-llm-alignment-via","title":"Less is More: Improving LLM Alignment via Preference Data Selection","date":"2025-02-20","arxiv_id":"2502.14560","n_code_links":0,"syntology":null},{"paper":null,"slug":"scaling-text-rich-image-understanding-via","title":"Scaling Text-Rich Image Understanding via Code-Guided Synthetic Multimodal Data Generation","date":"2025-02-20","arxiv_id":"2502.14846","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairkv-balancing-per-head-kv-cache-for-fast","title":"FairKV: Balancing Per-Head KV Cache for Fast Multi-GPU Inference","date":"2025-02-19","arxiv_id":"2502.15804","n_code_links":0,"syntology":null},{"paper":null,"slug":"giving-ai-personalities-leads-to-more-human","title":"Giving AI Personalities Leads to More Human-Like Reasoning","date":"2025-02-19","arxiv_id":"2502.14155","n_code_links":0,"syntology":null},{"paper":null,"slug":"theoretical-physics-benchmark-tpbench-a","title":"Theoretical Physics Benchmark (TPBench) -- a Dataset and Study of AI Reasoning Capabilities in Theoretical Physics","date":"2025-02-19","arxiv_id":"2502.15815","n_code_links":0,"syntology":null},{"paper":"/paper/thinkguard-deliberative-slow-thinking-leads","slug":"thinkguard-deliberative-slow-thinking-leads","title":"ThinkGuard: Deliberative Slow Thinking Leads to Cautious Guardrails","date":"2025-02-19","arxiv_id":"2502.13458","n_code_links":1,"syntology":null},{"paper":null,"slug":"dsmoe-matrix-partitioned-experts-with-dynamic","title":"DSMoE: Matrix-Partitioned Experts with Dynamic Routing for Computation-Efficient Dense LLMs","date":"2025-02-18","arxiv_id":"2502.12455","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-the-impact-of-quantization","slug":"investigating-the-impact-of-quantization","title":"Investigating the Impact of Quantization Methods on the Safety and Reliability of Large Language Models","date":"2025-02-18","arxiv_id":"2502.15799","n_code_links":1,"syntology":null},{"paper":"/paper/k-paths-reasoning-over-graph-paths-for-drug","slug":"k-paths-reasoning-over-graph-paths-for-drug","title":"K-Paths: Reasoning over Graph Paths for Drug Repurposing and Drug Interaction Prediction","date":"2025-02-18","arxiv_id":"2502.13344","n_code_links":1,"syntology":null},{"paper":null,"slug":"occult-evaluating-large-language-models-for","title":"OCCULT: Evaluating Large Language Models for Offensive Cyber Operation Capabilities","date":"2025-02-18","arxiv_id":"2502.15797","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-privacy-utility-and-efficiency","title":"Revisiting Privacy, Utility, and Efficiency Trade-offs when Fine-Tuning Large Language Models","date":"2025-02-18","arxiv_id":"2502.13313","n_code_links":0,"syntology":null},{"paper":"/paper/testing-prompt-engineering-methods-for","slug":"testing-prompt-engineering-methods-for","title":"Testing Prompt Engineering Methods for Knowledge Extraction from Text","date":"2025-02-18","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"independence-tests-for-language-models","title":"Independence Tests for Language Models","date":"2025-02-17","arxiv_id":"2502.12292","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-on-the-line-data-determines-loss-to-loss","title":"LLMs on the Line: Data Determines Loss-to-Loss Scaling Laws","date":"2025-02-17","arxiv_id":"2502.12120","n_code_links":0,"syntology":null},{"paper":"/paper/logic-py-bridging-the-gap-between-llms-and","slug":"logic-py-bridging-the-gap-between-llms-and","title":"Logic.py: Bridging the Gap between LLMs and Constraint Solvers","date":"2025-02-17","arxiv_id":"2502.15776","n_code_links":1,"syntology":null},{"paper":null,"slug":"smartllm-smart-contract-auditing-using-custom","title":"SmartLLM: Smart Contract Auditing using Custom Generative AI","date":"2025-02-17","arxiv_id":"2502.13167","n_code_links":0,"syntology":null},{"paper":"/paper/step-audio-unified-understanding-and","slug":"step-audio-unified-understanding-and","title":"Step-Audio: Unified Understanding and Generation in Intelligent Speech Interaction","date":"2025-02-17","arxiv_id":"2502.11946","n_code_links":1,"syntology":null},{"paper":"/paper/cola-compute-efficient-pre-training-of-llms","slug":"cola-compute-efficient-pre-training-of-llms","title":"CoLA: Compute-Efficient Pre-Training of LLMs via Low-Rank Activation","date":"2025-02-16","arxiv_id":"2502.10940","n_code_links":1,"syntology":null}],"record_sha256":"342568933aca5750894a42d3c038c816678be77801874de732b4641e699f306f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}