{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/llama/papers/5","list_of":"/method/llama","method":"LLaMA","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":11,"rows_per_page":100,"rows":[401,500],"of":1062,"counts":{"archive_papers_tagged":1062,"with_a_code_link":423,"where_syntology_ran_a_sample":143,"not_listed_spam_title":0,"listed":1062,"listed_where_code_ran":143,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":127,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":127,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/llama","prev":"/method/llama/papers/4","next":"/method/llama/papers/6","papers":[{"paper":"/paper/mllm-sul-multimodal-large-language-model-for","slug":"mllm-sul-multimodal-large-language-model-for","title":"MLLM-SUL: Multimodal Large Language Model for Semantic Scene Understanding and Localization in Traffic Scenarios","date":"2024-12-27","arxiv_id":"2412.19406","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-skill-adaptation-for-large-language","title":"Dynamic Skill Adaptation for Large Language Models","date":"2024-12-26","arxiv_id":"2412.19361","n_code_links":0,"syntology":null},{"paper":null,"slug":"whose-morality-do-they-speak-unraveling","title":"Whose Morality Do They Speak? Unraveling Cultural Bias in Multilingual Language Models","date":"2024-12-25","arxiv_id":"2412.18863","n_code_links":0,"syntology":null},{"paper":null,"slug":"genai-content-detection-task-2-ai-vs-human","title":"GenAI Content Detection Task 2: AI vs. Human -- Academic Essay Authenticity Challenge","date":"2024-12-24","arxiv_id":"2412.18274","n_code_links":0,"syntology":null},{"paper":"/paper/segment-based-attention-masking-for-gpts","slug":"segment-based-attention-masking-for-gpts","title":"Segment-Based Attention Masking for GPTs","date":"2024-12-24","arxiv_id":"2412.18487","n_code_links":1,"syntology":null},{"paper":null,"slug":"slimgpt-layer-wise-structured-pruning-for","title":"SlimGPT: Layer-wise Structured Pruning for Large Language Models","date":"2024-12-24","arxiv_id":"2412.18110","n_code_links":0,"syntology":null},{"paper":null,"slug":"gqsa-group-quantization-and-sparsity-for","title":"GQSA: Group Quantization and Sparsity for Accelerating Large Language Model Inference","date":"2024-12-23","arxiv_id":"2412.17560","n_code_links":0,"syntology":null},{"paper":"/paper/highly-optimized-kernels-and-fine-grained","slug":"highly-optimized-kernels-and-fine-grained","title":"Highly Optimized Kernels and Fine-Grained Codebooks for LLM Inference on Arm CPUs","date":"2024-12-23","arxiv_id":"2501.00032","n_code_links":1,"syntology":null},{"paper":"/paper/lmv-rpa-large-model-voting-based-robotic","slug":"lmv-rpa-large-model-voting-based-robotic","title":"LMV-RPA: Large Model Voting-based Robotic Process Automation","date":"2024-12-23","arxiv_id":"2412.17965","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-llm-reasoning-in-the-operations","slug":"evaluating-llm-reasoning-in-the-operations","title":"Evaluating LLM Reasoning in the Operations Research Domain with ORQA","date":"2024-12-22","arxiv_id":"2412.17874","n_code_links":2,"syntology":null},{"paper":"/paper/joint-knowledge-editing-for-information","slug":"joint-knowledge-editing-for-information","title":"Joint Knowledge Editing for Information Enrichment and Probability Promotion","date":"2024-12-22","arxiv_id":"2412.17872","n_code_links":1,"syntology":null},{"paper":"/paper/psychadapter-adapting-llm-transformers-to","slug":"psychadapter-adapting-llm-transformers-to","title":"PsychAdapter: Adapting LLM Transformers to Reflect Traits, Personality and Mental Health","date":"2024-12-22","arxiv_id":"2412.16882","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-the-performance-of-large-language-4","title":"Evaluating the Performance of Large Language Models in Scientific Claim Detection and Classification","date":"2024-12-21","arxiv_id":"2412.16486","n_code_links":0,"syntology":null},{"paper":null,"slug":"infotech-assistant-a-multimodal","title":"InfoTech Assistant : A Multimodal Conversational Agent for InfoTechnology Web Portal Queries","date":"2024-12-21","arxiv_id":"2412.16412","n_code_links":0,"syntology":null},{"paper":"/paper/silvar-speech-driven-multimodal-model-for","slug":"silvar-speech-driven-multimodal-model-for","title":"SilVar: Speech Driven Multimodal Model for Reasoning Visual Question Answering and Object Localization","date":"2024-12-21","arxiv_id":"2412.16771","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-machine-learning-approach-for-emergency","title":"A Machine Learning Approach for Emergency Detection in Medical Scenarios Using Large Language Models","date":"2024-12-20","arxiv_id":"2412.16341","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-obfuscate-code-a-systematic-analysis","title":"Can LLMs Obfuscate Code? A Systematic Analysis of Large Language Models into Assembly Code Obfuscation","date":"2024-12-20","arxiv_id":"2412.16135","n_code_links":0,"syntology":null},{"paper":"/paper/conflibert-a-language-model-for-political","slug":"conflibert-a-language-model-for-political","title":"ConfliBERT: A Language Model for Political Conflict","date":"2024-12-19","arxiv_id":"2412.15060","n_code_links":1,"syntology":null},{"paper":null,"slug":"directorllm-for-human-centric-video","title":"DirectorLLM for Human-Centric Video Generation","date":"2024-12-19","arxiv_id":"2412.14484","n_code_links":0,"syntology":null},{"paper":null,"slug":"mixllm-llm-quantization-with-global-mixed","title":"MixLLM: LLM Quantization with Global Mixed-precision between Output-features and Highly-efficient System Design","date":"2024-12-19","arxiv_id":"2412.14590","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-the-dark-side-of-llms-intrinsic","title":"Understanding the Dark Side of LLMs' Intrinsic Self-Correction","date":"2024-12-19","arxiv_id":"2412.14959","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-adaptative-continual-learning-for-low","title":"Domain-adaptative Continual Learning for Low-resource Tasks: Evaluation on Nepali","date":"2024-12-18","arxiv_id":"2412.13860","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-knowledge-distillation-for-llms","slug":"enhancing-knowledge-distillation-for-llms","title":"Enhancing Knowledge Distillation for LLMs with Response-Priming Prompting","date":"2024-12-18","arxiv_id":"2412.17846","n_code_links":1,"syntology":null},{"paper":null,"slug":"lift-improving-long-context-understanding","title":"LIFT: Improving Long Context Understanding Through Long Input Fine-Tuning","date":"2024-12-18","arxiv_id":"2412.13626","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-sem-a-sentiment-based-student-engagement","title":"LLM-SEM: A Sentiment-Based Student Engagement Metric Using LLMS for E-Learning Platforms","date":"2024-12-18","arxiv_id":"2412.13765","n_code_links":0,"syntology":null},{"paper":"/paper/mix-ln-unleashing-the-power-of-deeper-layers","slug":"mix-ln-unleashing-the-power-of-deeper-layers","title":"Mix-LN: Unleashing the Power of Deeper Layers by Combining Pre-LN and Post-LN","date":"2024-12-18","arxiv_id":"2412.13795","n_code_links":1,"syntology":null},{"paper":"/paper/resq-mixed-precision-quantization-of-large","slug":"resq-mixed-precision-quantization-of-large","title":"ResQ: Mixed-Precision Quantization of Large Language Models with Low-Rank Residuals","date":"2024-12-18","arxiv_id":"2412.14363","n_code_links":1,"syntology":null},{"paper":"/paper/typhoon-2-a-family-of-open-text-and","slug":"typhoon-2-a-family-of-open-text-and","title":"Typhoon 2: A Family of Open Text and Multimodal Thai Large Language Models","date":"2024-12-18","arxiv_id":"2412.13702","n_code_links":1,"syntology":null},{"paper":"/paper/algorithmic-fidelity-of-large-language-models","slug":"algorithmic-fidelity-of-large-language-models","title":"Algorithmic Fidelity of Large Language Models in Generating Synthetic German Public Opinions: A Case Study","date":"2024-12-17","arxiv_id":"2412.13169","n_code_links":1,"syntology":null},{"paper":"/paper/extending-llms-to-new-languages-a-case-study","slug":"extending-llms-to-new-languages-a-case-study","title":"Extending LLMs to New Languages: A Case Study of Llama and Persian Adaptation","date":"2024-12-17","arxiv_id":"2412.13375","n_code_links":1,"syntology":null},{"paper":null,"slug":"llmcl-gec-advancing-grammatical-error","title":"LLMCL-GEC: Advancing Grammatical Error Correction with LLM-Driven Curriculum Learning","date":"2024-12-17","arxiv_id":"2412.12541","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-are-also-effective-embedding-models-an","title":"LLMs are Also Effective Embedding Models: An In-depth Overview","date":"2024-12-17","arxiv_id":"2412.12591","n_code_links":0,"syntology":null},{"paper":null,"slug":"swan-preprocessing-sgd-enables-adam-level","title":"SWAN: SGD with Normalization and Whitening Enables Stateless LLM Training","date":"2024-12-17","arxiv_id":"2412.13148","n_code_links":0,"syntology":null},{"paper":"/paper/causal-diffusion-transformers-for-generative","slug":"causal-diffusion-transformers-for-generative","title":"Causal Diffusion Transformers for Generative Modeling","date":"2024-12-16","arxiv_id":"2412.12095","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":6,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["causalfusion/causalfusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"codenames-as-a-benchmark-for-large-language","title":"Codenames as a Benchmark for Large Language Models","date":"2024-12-16","arxiv_id":"2412.11373","n_code_links":0,"syntology":null},{"paper":"/paper/combining-large-language-models-with-tutoring","slug":"combining-large-language-models-with-tutoring","title":"Combining Large Language Models with Tutoring System Intelligence: A Case Study in Caregiver Homework Support","date":"2024-12-16","arxiv_id":"2412.11995","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-open-source-advantage-in-large-language","title":"The Open Source Advantage in Large Language Models (LLMs)","date":"2024-12-16","arxiv_id":"2412.12004","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-discrete-personas-personality-modeling","slug":"beyond-discrete-personas-personality-modeling","title":"Beyond Discrete Personas: Personality Modeling Through Journal Intensive Conversations","date":"2024-12-15","arxiv_id":"2412.11250","n_code_links":1,"syntology":null},{"paper":null,"slug":"sequence-level-analysis-of-leakage-risk-of","title":"Sequence-Level Leakage Risk of Training Data in Large Language Models","date":"2024-12-15","arxiv_id":"2412.11302","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-recent-evaluation-on-the-performance-of","title":"A recent evaluation on the performance of LLMs on radiation oncology physics using questions of randomly shuffled options","date":"2024-12-14","arxiv_id":"2412.10622","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-vehicle-plate-recognition","title":"Advancing Vehicle Plate Recognition: Multitasking Visual Language Models with VehiclePaliGemma","date":"2024-12-14","arxiv_id":"2412.14197","n_code_links":0,"syntology":null},{"paper":"/paper/do-large-language-vision-models-understand-3d","slug":"do-large-language-vision-models-understand-3d","title":"Do large language vision models understand 3D shapes?","date":"2024-12-14","arxiv_id":"2412.10908","n_code_links":1,"syntology":null},{"paper":"/paper/llama-3-meets-moe-efficient-upcycling","slug":"llama-3-meets-moe-efficient-upcycling","title":"Llama 3 Meets MoE: Efficient Upcycling","date":"2024-12-13","arxiv_id":"2412.09952","n_code_links":1,"syntology":null},{"paper":"/paper/mst-r-multi-stage-tuning-for-retrieval","slug":"mst-r-multi-stage-tuning-for-retrieval","title":"MST-R: Multi-Stage Tuning for Retrieval Systems and Metric Evaluation","date":"2024-12-13","arxiv_id":"2412.10313","n_code_links":1,"syntology":null},{"paper":null,"slug":"targeted-angular-reversal-of-weights-tars-for","title":"Targeted Angular Reversal of Weights (TARS) for Knowledge Removal in Large Language Models","date":"2024-12-13","arxiv_id":"2412.10257","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-llms-for-mimicking-child","title":"Benchmarking LLMs for Mimicking Child-Caregiver Language in Interaction","date":"2024-12-12","arxiv_id":"2412.09318","n_code_links":0,"syntology":null},{"paper":"/paper/foundational-large-language-models-for","slug":"foundational-large-language-models-for","title":"Foundational Large Language Models for Materials Research","date":"2024-12-12","arxiv_id":"2412.09560","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":4,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["M3RG-IITD/llamat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/lexico-extreme-kv-cache-compression-via","slug":"lexico-extreme-kv-cache-compression-via","title":"Lexico: Extreme KV Cache Compression via Sparse Coding over Universal Dictionaries","date":"2024-12-12","arxiv_id":"2412.08890","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-personalized-ai-mentoring-with","title":"Assessing Personalized AI Mentoring with Large Language Models in the Computing Field","date":"2024-12-11","arxiv_id":"2412.08430","n_code_links":0,"syntology":null},{"paper":"/paper/nyayaanumana-inlegalllama-the-largest-indian","slug":"nyayaanumana-inlegalllama-the-largest-indian","title":"NyayaAnumana & INLegalLlama: The Largest Indian Legal Judgment Prediction Dataset and Specialized Language Model for Enhanced Decision Analysis","date":"2024-12-11","arxiv_id":"2412.08385","n_code_links":1,"syntology":null},{"paper":"/paper/frame-representation-hypothesis-multi-token","slug":"frame-representation-hypothesis-multi-token","title":"Frame Representation Hypothesis: Multi-Token LLM Interpretability and Concept-Guided Text Generation","date":"2024-12-10","arxiv_id":"2412.07334","n_code_links":1,"syntology":null},{"paper":null,"slug":"generating-knowledge-graphs-from-large","title":"Generating Knowledge Graphs from Large Language Models: A Comparative Study of GPT-4, LLaMA 2, and BERT","date":"2024-12-10","arxiv_id":"2412.07412","n_code_links":0,"syntology":null},{"paper":"/paper/identifying-and-manipulating-personality","slug":"identifying-and-manipulating-personality","title":"Identifying and Manipulating Personality Traits in LLMs Through Activation Engineering","date":"2024-12-10","arxiv_id":"2412.10427","n_code_links":1,"syntology":null},{"paper":null,"slug":"memhunter-automated-and-verifiable","title":"MemHunter: Automated and Verifiable Memorization Detection at Dataset-scale in LLMs","date":"2024-12-10","arxiv_id":"2412.07261","n_code_links":0,"syntology":null},{"paper":null,"slug":"trojanwhisper-evaluating-pre-trained-llms-to","title":"TrojanWhisper: Evaluating Pre-trained LLMs to Detect and Localize Hardware Trojans","date":"2024-12-10","arxiv_id":"2412.07636","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-atc-coding-with-large-language","title":"Zero-Shot ATC Coding with Large Language Models for Clinical Assessments","date":"2024-12-10","arxiv_id":"2412.07743","n_code_links":0,"syntology":null},{"paper":null,"slug":"supermerge-an-approach-for-gradient-based","title":"SUPERMERGE: An Approach For Gradient-Based Model Merging","date":"2024-12-09","arxiv_id":"2412.10416","n_code_links":0,"syntology":null},{"paper":null,"slug":"domain-specific-translation-with-open-source","title":"Domain-Specific Translation with Open-Source Large Language Models: Resource-Oriented Analysis","date":"2024-12-08","arxiv_id":"2412.05862","n_code_links":0,"syntology":null},{"paper":"/paper/fully-open-source-moxin-7b-technical-report","slug":"fully-open-source-moxin-7b-technical-report","title":"Fully Open Source Moxin-7B Technical Report","date":"2024-12-08","arxiv_id":"2412.06845","n_code_links":1,"syntology":null},{"paper":null,"slug":"taming-sensitive-weights-noise-perturbation","title":"Taming Sensitive Weights : Noise Perturbation Fine-tuning for Robust LLM Quantization","date":"2024-12-08","arxiv_id":"2412.06858","n_code_links":0,"syntology":null},{"paper":null,"slug":"comprehensive-evaluation-of-multimodal-ai","title":"Comprehensive Evaluation of Multimodal AI Models in Medical Imaging Diagnosis: From Data Augmentation to Preference-Based Comparison","date":"2024-12-07","arxiv_id":"2412.05536","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatnvd-advancing-cybersecurity-vulnerability","title":"ChatNVD: Advancing Cybersecurity Vulnerability Assessment with Large Language Models","date":"2024-12-06","arxiv_id":"2412.04756","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-cross-language-code-translation-via","title":"Enhancing Cross-Language Code Translation via Task-Specific Embedding Alignment in Retrieval-Augmented Generation","date":"2024-12-06","arxiv_id":"2412.05159","n_code_links":0,"syntology":null},{"paper":null,"slug":"frontier-models-are-capable-of-in-context","title":"Frontier Models are Capable of In-context Scheming","date":"2024-12-06","arxiv_id":"2412.04984","n_code_links":0,"syntology":null},{"paper":null,"slug":"aya-expanse-combining-research-breakthroughs","title":"Aya Expanse: Combining Research Breakthroughs for a New Multilingual Frontier","date":"2024-12-05","arxiv_id":"2412.04261","n_code_links":0,"syntology":null},{"paper":"/paper/extractive-structures-learned-in-pretraining","slug":"extractive-structures-learned-in-pretraining","title":"Extractive Structures Learned in Pretraining Enable Generalization on Finetuned Facts","date":"2024-12-05","arxiv_id":"2412.04614","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jiahai-feng/extractive-structures"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/florence-vl-enhancing-vision-language-models","slug":"florence-vl-enhancing-vision-language-models","title":"Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion","date":"2024-12-05","arxiv_id":"2412.04424","n_code_links":1,"syntology":null},{"paper":null,"slug":"skim-any-bit-quantization-pushing-the-limits","title":"SKIM: Any-bit Quantization Pushing The Limits of Post-Training Quantization","date":"2024-12-05","arxiv_id":"2412.04180","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-gender-bias-transfer-between-pre","title":"Evaluating Gender Bias Transfer between Pre-trained and Prompt-Adapted Language Models","date":"2024-12-04","arxiv_id":"2412.03537","n_code_links":0,"syntology":null},{"paper":"/paper/from-language-models-over-tokens-to-language","slug":"from-language-models-over-tokens-to-language","title":"From Language Models over Tokens to Language Models over Characters","date":"2024-12-04","arxiv_id":"2412.03719","n_code_links":0,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"kklip-knowledge-distillation-exploiting-k","title":"Enhancing CLIP Conceptual Embedding through Knowledge Distillation","date":"2024-12-04","arxiv_id":"2412.03513","n_code_links":0,"syntology":null},{"paper":null,"slug":"cegi-measuring-the-trade-off-between","title":"CEGI: Measuring the trade-off between efficiency and carbon emissions for SLMs and VLMs","date":"2024-12-03","arxiv_id":"2412.02602","n_code_links":0,"syntology":null},{"paper":null,"slug":"compressing-kv-cache-for-long-context-llm","title":"Compressing KV Cache for Long-Context LLM Inference with Inter-Layer Attention Similarity","date":"2024-12-03","arxiv_id":"2412.02252","n_code_links":0,"syntology":null},{"paper":null,"slug":"nemotron-cc-transforming-common-crawl-into-a","title":"Nemotron-CC: Transforming Common Crawl into a Refined Long-Horizon Pretraining Dataset","date":"2024-12-03","arxiv_id":"2412.02595","n_code_links":0,"syntology":null},{"paper":"/paper/rare-retrieval-augmented-reasoning","slug":"rare-retrieval-augmented-reasoning","title":"RARE: Retrieval-Augmented Reasoning Enhancement for Large Language Models","date":"2024-12-03","arxiv_id":"2412.02830","n_code_links":1,"syntology":null},{"paper":null,"slug":"early-exit-is-a-natural-capability-in","title":"Early Exit Is a Natural Capability in Transformer-based Models: An Empirical Study on Early Exit without Joint Optimization","date":"2024-12-02","arxiv_id":"2412.01455","n_code_links":0,"syntology":null},{"paper":null,"slug":"malt-improving-reasoning-with-multi-agent-llm","title":"MALT: Improving Reasoning with Multi-Agent LLM Training","date":"2024-12-02","arxiv_id":"2412.01928","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-medical-disease-classification","title":"Multimodal Medical Disease Classification with LLaMA II","date":"2024-12-02","arxiv_id":"2412.01306","n_code_links":0,"syntology":null},{"paper":null,"slug":"uhura-a-benchmark-for-evaluating-scientific","title":"Uhura: A Benchmark for Evaluating Scientific Question Answering and Truthfulness in Low-Resource African Languages","date":"2024-12-01","arxiv_id":"2412.00948","n_code_links":0,"syntology":null},{"paper":null,"slug":"llama-gene-a-general-purpose-gene-task-large","title":"LLaMA-Gene: A General-purpose Gene Task Large Language Model Based on Instruction Fine-tuning","date":"2024-11-30","arxiv_id":"2412.00471","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-review-of-llm-based-explanations-in","title":"On Explaining Recommendations with Large Language Models: A Review","date":"2024-11-29","arxiv_id":"2411.19576","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-large-language-models-to-deliver","slug":"fine-tuning-large-language-models-to-deliver","title":"Fine Tuning Large Language Models to Deliver CBT for Depression","date":"2024-11-29","arxiv_id":"2412.00251","n_code_links":1,"syntology":null},{"paper":null,"slug":"sensitive-content-classification-in-social","title":"Sensitive Content Classification in Social Media: A Holistic Resource and Evaluation","date":"2024-11-29","arxiv_id":"2411.19832","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-extensive-evaluation-of-factual","title":"An Extensive Evaluation of Factual Consistency in Large Language Models for Data-to-Text Generation","date":"2024-11-28","arxiv_id":"2411.19203","n_code_links":0,"syntology":null},{"paper":"/paper/deniahl-in-context-features-influence-llm","slug":"deniahl-in-context-features-influence-llm","title":"DENIAHL: In-Context Features Influence LLM Needle-In-A-Haystack Abilities","date":"2024-11-28","arxiv_id":"2411.19360","n_code_links":1,"syntology":null},{"paper":null,"slug":"diesel-dynamic-inference-guidance-via-evasion","title":"DIESEL -- Dynamic Inference-Guidance via Evasion of Semantic Embeddings in LLMs","date":"2024-11-28","arxiv_id":"2411.19038","n_code_links":0,"syntology":null},{"paper":null,"slug":"sneaking-syntax-into-transformer-language","title":"Sneaking Syntax into Transformer Language Models with Tree Regularization","date":"2024-11-28","arxiv_id":"2411.18885","n_code_links":0,"syntology":null},{"paper":"/paper/emergence-of-self-identity-in-ai-a","slug":"emergence-of-self-identity-in-ai-a","title":"Emergence of Self-Identity in AI: A Mathematical Framework and Empirical Study with Generative Large Language Models","date":"2024-11-27","arxiv_id":"2411.18530","n_code_links":1,"syntology":null},{"paper":"/paper/training-and-evaluating-language-models-with","slug":"training-and-evaluating-language-models-with","title":"Training and Evaluating Language Models with Template-based Data Generation","date":"2024-11-27","arxiv_id":"2411.18104","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["iiis-ai/templatemath"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"advancing-content-moderation-evaluating-large","title":"Advancing Content Moderation: Evaluating Large Language Models for Detecting Sensitive Content Across Text, Images, and Videos","date":"2024-11-26","arxiv_id":"2411.17123","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-limitations-of-llm-as-annotator-for-low","title":"On Limitations of LLM as Annotator for Low Resource Languages","date":"2024-11-26","arxiv_id":"2411.17637","n_code_links":0,"syntology":null},{"paper":"/paper/bayling-2-a-multilingual-large-language-model","slug":"bayling-2-a-multilingual-large-language-model","title":"BayLing 2: A Multilingual Large Language Model with Efficient Language Alignment","date":"2024-11-25","arxiv_id":"2411.16300","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-ai-grade-your-essays-a-comparative","title":"Can AI grade your essays? A comparative analysis of large language models and teacher ratings in multidimensional essay scoring","date":"2024-11-25","arxiv_id":"2411.16337","n_code_links":0,"syntology":null},{"paper":"/paper/cautious-optimizers-improving-training-with","slug":"cautious-optimizers-improving-training-with","title":"Cautious Optimizers: Improving Training with One Line of Code","date":"2024-11-25","arxiv_id":"2411.16085","n_code_links":3,"syntology":{"ran":7,"of":12,"n_ran_checked":3,"n_instrument":4,"unverified":5,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["kyleliang919/c-optim"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dynamic-self-distillation-via-previous-mini","title":"Dynamic Self-Distillation via Previous Mini-batches for Fine-tuning Small Language Models","date":"2024-11-25","arxiv_id":"2411.16991","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-answer-reliability-through-inter","title":"Enhancing Answer Reliability Through Inter-Model Consensus of Large Language Models","date":"2024-11-25","arxiv_id":"2411.16797","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-two-hop-curse-llms-trained-on-a-b-b-c","title":"The Two-Hop Curse: LLMs trained on A$\\rightarrow$B, B$\\rightarrow$C fail to learn A$\\rightarrow$C","date":"2024-11-25","arxiv_id":"2411.16353","n_code_links":0,"syntology":null},{"paper":null,"slug":"anda-unlocking-efficient-llm-inference-with-a","title":"Anda: Unlocking Efficient LLM Inference with a Variable-Length Grouped Activation Data Format","date":"2024-11-24","arxiv_id":"2411.15982","n_code_links":0,"syntology":null},{"paper":"/paper/llama-moe-v2-exploring-sparsity-of-llama-from","slug":"llama-moe-v2-exploring-sparsity-of-llama-from","title":"LLaMA-MoE v2: Exploring Sparsity of LLaMA from Perspective of Mixture-of-Experts with Post-Training","date":"2024-11-24","arxiv_id":"2411.15708","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opensparsellms/llama-moe-v2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/moralized-multi-step-jailbreak-prompts-black","slug":"moralized-multi-step-jailbreak-prompts-black","title":"\"Moralized\" Multi-Step Jailbreak Prompts: Black-Box Testing of Guardrails in Large Language Models for Verbal Attacks","date":"2024-11-23","arxiv_id":"2411.16730","n_code_links":1,"syntology":null}],"record_sha256":"d356a3fc5966afab596a4bdb00272298c137e8312abfd6ae0dd6d32a9731a735","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}