{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/llama/papers/10","list_of":"/method/llama","method":"LLaMA","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":11,"rows_per_page":100,"rows":[901,1000],"of":1062,"counts":{"archive_papers_tagged":1062,"with_a_code_link":423,"where_syntology_ran_a_sample":143,"not_listed_spam_title":0,"listed":1062,"listed_where_code_ran":143,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":127,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":127,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/llama","prev":"/method/llama/papers/9","next":"/method/llama/papers/11","papers":[{"paper":"/paper/simpo-simple-preference-optimization-with-a","slug":"simpo-simple-preference-optimization-with-a","title":"SimPO: Simple Preference Optimization with a Reference-Free Reward","date":"2024-05-23","arxiv_id":"2405.14734","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":5,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princeton-nlp/simpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/wise-rethinking-the-knowledge-memory-for","slug":"wise-rethinking-the-knowledge-memory-for","title":"WISE: Rethinking the Knowledge Memory for Lifelong Model Editing of Large Language Models","date":"2024-05-23","arxiv_id":"2405.14768","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":4,"n_instrument":9,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cg-fedllm-how-to-compress-gradients-in","title":"CG-FedLLM: How to Compress Gradients in Federated Fune-tuning for Large Language Models","date":"2024-05-22","arxiv_id":"2405.13746","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-large-language-models-with-human","slug":"evaluating-large-language-models-with-human","title":"Evaluating Large Language Models with Human Feedback: Establishing a Swedish Benchmark","date":"2024-05-22","arxiv_id":"2405.14006","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-to-set-adamw-s-weight-decay-as-you-scale","title":"How to set AdamW's weight decay as you scale model and dataset size","date":"2024-05-22","arxiv_id":"2405.13698","n_code_links":0,"syntology":null},{"paper":"/paper/semantic-density-uncertainty-quantification","slug":"semantic-density-uncertainty-quantification","title":"Semantic Density: Uncertainty Quantification for Large Language Models through Confidence Measurement in Semantic Space","date":"2024-05-22","arxiv_id":"2405.13845","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":2,"n_instrument":3,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cognizant-ai-labs/semantic-density-paper"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-ai-and-large-language-models-for","title":"Generative AI in Cybersecurity: A Comprehensive Review of LLM Applications and Vulnerabilities","date":"2024-05-21","arxiv_id":"2405.12750","n_code_links":0,"syntology":null},{"paper":"/paper/your-transformer-is-secretly-linear","slug":"your-transformer-is-secretly-linear","title":"Your Transformer is Secretly Linear","date":"2024-05-19","arxiv_id":"2405.12250","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AIRI-Institute/LLM-Microscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"activellm-large-language-model-based-active","title":"ActiveLLM: Large Language Model-based Active Learning for Textual Few-Shot Scenarios","date":"2024-05-17","arxiv_id":"2405.10808","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-large-language-models-moral-hypocrites-a","title":"Are Large Language Models Moral Hypocrites? A Study Based on Moral Foundations","date":"2024-05-17","arxiv_id":"2405.11100","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-dialogue-state-tracking-models","title":"Enhancing Dialogue State Tracking Models through LLM-backed User-Agents Simulation","date":"2024-05-17","arxiv_id":"2405.13037","n_code_links":0,"syntology":null},{"paper":"/paper/documint-docstring-generation-for-python","slug":"documint-docstring-generation-for-python","title":"DocuMint: Docstring Generation for Python using Small Language Models","date":"2024-05-16","arxiv_id":"2405.10243","n_code_links":1,"syntology":null},{"paper":"/paper/many-shot-in-context-learning-in-multimodal","slug":"many-shot-in-context-learning-in-multimodal","title":"Many-Shot In-Context Learning in Multimodal Foundation Models","date":"2024-05-16","arxiv_id":"2405.09798","n_code_links":1,"syntology":null},{"paper":null,"slug":"dynamic-activation-pitfalls-in-llama-models","title":"Dynamic Activation Pitfalls in LLaMA Models: An Empirical Study","date":"2024-05-15","arxiv_id":"2405.09274","n_code_links":0,"syntology":null},{"paper":"/paper/coding-historical-causes-of-death-data-with","slug":"coding-historical-causes-of-death-data-with","title":"Coding historical causes of death data with Large Language Models","date":"2024-05-13","arxiv_id":"2405.07560","n_code_links":1,"syntology":null},{"paper":"/paper/macbehaviour-an-r-package-for-behavioural","slug":"macbehaviour-an-r-package-for-behavioural","title":"MacBehaviour: An R package for behavioural experimentation on large language models","date":"2024-05-13","arxiv_id":"2405.07495","n_code_links":1,"syntology":null},{"paper":"/paper/parden-can-you-repeat-that-defending-against","slug":"parden-can-you-repeat-that-defending-against","title":"PARDEN, Can You Repeat That? Defending against Jailbreaks via Repetition","date":"2024-05-13","arxiv_id":"2405.07932","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":3,"n_instrument":3,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ed-zh/parden"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"uccix-irish-excellence-large-language-model","title":"UCCIX: Irish-eXcellence Large Language Model","date":"2024-05-13","arxiv_id":"2405.13010","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-multiple-post-training-techniques","title":"Post Training Quantization of Large Language Models with Microscaling Formats","date":"2024-05-12","arxiv_id":"2405.07135","n_code_links":0,"syntology":null},{"paper":null,"slug":"automating-thematic-analysis-how-llms-analyse","title":"Automating Thematic Analysis: How LLMs Analyse Controversial Topics","date":"2024-05-11","arxiv_id":"2405.06919","n_code_links":0,"syntology":null},{"paper":null,"slug":"canal-cyber-activity-news-alerting-language","title":"CANAL -- Cyber Activity News Alerting Language Model: Empirical Approach vs. Expensive LLM","date":"2024-05-10","arxiv_id":"2405.06772","n_code_links":0,"syntology":null},{"paper":null,"slug":"characterizing-the-accuracy-efficiency-trade","title":"Characterizing the Accuracy -- Efficiency Trade-off of Low-rank Decomposition in Language Models","date":"2024-05-10","arxiv_id":"2405.06626","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-show-human-like-social","title":"Large Language Models Show Human-like Social Desirability Biases in Survey Responses","date":"2024-05-09","arxiv_id":"2405.06058","n_code_links":0,"syntology":null},{"paper":null,"slug":"information-extraction-from-historical-well","title":"Information Extraction from Historical Well Records Using A Large Language Model","date":"2024-05-08","arxiv_id":"2405.05438","n_code_links":0,"syntology":null},{"paper":null,"slug":"kv-runahead-scalable-causal-llm-inference-by","title":"KV-Runahead: Scalable Causal LLM Inference by Parallel Key-Value Cache Generation","date":"2024-05-08","arxiv_id":"2405.05329","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-as-a-metacognitive-agent-a","title":"Generative AI as a metacognitive agent: A comparative mixed-method study with human participants on ICF-mimicking exam performance","date":"2024-05-07","arxiv_id":"2405.05285","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-controlled-experiment-on-the-energy","title":"A Controlled Experiment on the Energy Efficiency of the Source Code Generated by Code Llama","date":"2024-05-06","arxiv_id":"2405.03616","n_code_links":0,"syntology":null},{"paper":"/paper/anchored-answers-unravelling-positional-bias","slug":"anchored-answers-unravelling-positional-bias","title":"Anchored Answers: Unravelling Positional Bias in GPT-2's Multiple-Choice Questions","date":"2024-05-06","arxiv_id":"2405.03205","n_code_links":1,"syntology":null},{"paper":null,"slug":"enabling-high-sparsity-foundational-llama","title":"Enabling High-Sparsity Foundational Llama Models with Efficient Pretraining and Deployment","date":"2024-05-06","arxiv_id":"2405.03594","n_code_links":0,"syntology":null},{"paper":null,"slug":"quantifying-the-capabilities-of-llms-across","title":"Quantifying the Capabilities of LLMs across Scale and Precision","date":"2024-05-06","arxiv_id":"2405.03146","n_code_links":0,"syntology":null},{"paper":null,"slug":"wdmoe-wireless-distributed-large-language","title":"WDMoE: Wireless Distributed Large Language Models with Mixture of Experts","date":"2024-05-06","arxiv_id":"2405.03131","n_code_links":0,"syntology":null},{"paper":null,"slug":"iceformer-accelerated-inference-with-long","title":"IceFormer: Accelerated Inference with Long-Sequence Transformers on CPUs","date":"2024-05-05","arxiv_id":"2405.02842","n_code_links":0,"syntology":null},{"paper":"/paper/negativeprompt-leveraging-psychology-for","slug":"negativeprompt-leveraging-psychology-for","title":"NegativePrompt: Leveraging Psychology for Large Language Models Enhancement via Negative Emotional Stimuli","date":"2024-05-05","arxiv_id":"2405.02814","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wangxu0820/negativeprompt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/aloe-a-family-of-fine-tuned-open-healthcare","slug":"aloe-a-family-of-fine-tuned-open-healthcare","title":"Aloe: A Family of Fine-tuned Open Healthcare LLMs","date":"2024-05-03","arxiv_id":"2405.01886","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":null}},{"paper":null,"slug":"evaluating-large-language-models-for-2","title":"Evaluating Large Language Models for Structured Science Summarization in the Open Research Knowledge Graph","date":"2024-05-03","arxiv_id":"2405.02105","n_code_links":0,"syntology":null},{"paper":"/paper/nemo-aligner-scalable-toolkit-for-efficient","slug":"nemo-aligner-scalable-toolkit-for-efficient","title":"NeMo-Aligner: Scalable Toolkit for Efficient Model Alignment","date":"2024-05-02","arxiv_id":"2405.01481","n_code_links":1,"syntology":null},{"paper":"/paper/better-faster-large-language-models-via-multi","slug":"better-faster-large-language-models-via-multi","title":"Better & Faster Large Language Models via Multi-token Prediction","date":"2024-04-30","arxiv_id":"2404.19737","n_code_links":1,"syntology":null},{"paper":"/paper/analyzing-semantic-change-through-lexical","slug":"analyzing-semantic-change-through-lexical","title":"Analyzing Semantic Change through Lexical Replacements","date":"2024-04-29","arxiv_id":"2404.18570","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["changeiskey/asc-lr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hlstransform-energy-efficient-llama-2","slug":"hlstransform-energy-efficient-llama-2","title":"HLSTransform: Energy-Efficient Llama 2 Inference on FPGAs Via High Level Synthesis","date":"2024-04-29","arxiv_id":"2405.00738","n_code_links":1,"syntology":null},{"paper":"/paper/markovian-agents-for-truthful-language","slug":"markovian-agents-for-truthful-language","title":"Markovian Transformers for Informative Language Modeling","date":"2024-04-29","arxiv_id":"2404.18988","n_code_links":1,"syntology":{"ran":15,"of":16,"n_ran_checked":15,"n_instrument":0,"unverified":1,"pointer_only":16,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 1 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["scottviteri/markoviantraining"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"parameter-efficient-tuning-large-language","title":"Parameter-Efficient Tuning Large Language Models for Graph Representation Learning","date":"2024-04-28","arxiv_id":"2404.18271","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-a-large-japanese-web-corpus-for","title":"Building a Large Japanese Web Corpus for Large Language Models","date":"2024-04-27","arxiv_id":"2404.17733","n_code_links":0,"syntology":null},{"paper":null,"slug":"continual-pre-training-for-cross-lingual-llm","title":"Continual Pre-Training for Cross-Lingual LLM Adaptation: Enhancing Japanese Language Capabilities","date":"2024-04-27","arxiv_id":"2404.17790","n_code_links":0,"syntology":null},{"paper":"/paper/llmparser-an-exploratory-study-on-using-large","slug":"llmparser-an-exploratory-study-on-using-large","title":"LLMParser: An Exploratory Study on Using Large Language Models for Log Parsing","date":"2024-04-27","arxiv_id":"2404.18001","n_code_links":1,"syntology":null},{"paper":"/paper/indicgenbench-a-multilingual-benchmark-to","slug":"indicgenbench-a-multilingual-benchmark-to","title":"IndicGenBench: A Multilingual Benchmark to Evaluate Generation Capabilities of LLMs on Indic Languages","date":"2024-04-25","arxiv_id":"2404.16816","n_code_links":1,"syntology":null},{"paper":"/paper/layer-skip-enabling-early-exit-inference-and","slug":"layer-skip-enabling-early-exit-inference-and","title":"LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding","date":"2024-04-25","arxiv_id":"2404.16710","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":1,"n_instrument":3,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/layerskip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"assessing-the-potential-of-mid-sized-language","title":"Assessing The Potential Of Mid-Sized Language Models For Clinical QA","date":"2024-04-24","arxiv_id":"2404.15894","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-instruction-tuning-make-llms-more","title":"Does Instruction Tuning Make LLMs More Consistent?","date":"2024-04-23","arxiv_id":"2404.15206","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-enhanced-language-models-for","title":"Context-Enhanced Language Models for Generating Multi-Paper Citations","date":"2024-04-22","arxiv_id":"2404.13865","n_code_links":0,"syntology":null},{"paper":null,"slug":"expert-router-orchestrating-efficient","title":"Performance Characterization of Expert Router for Scalable LLM Inference","date":"2024-04-22","arxiv_id":"2404.15153","n_code_links":0,"syntology":null},{"paper":"/paper/how-good-are-low-bit-quantized-llama3-models","slug":"how-good-are-low-bit-quantized-llama3-models","title":"An empirical study of LLaMA3 quantization: from LLMs to MLLMs","date":"2024-04-22","arxiv_id":"2404.14047","n_code_links":2,"syntology":{"ran":7,"of":10,"n_ran_checked":4,"n_instrument":3,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["macaronlin/llama3-quantization"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/phi-3-technical-report-a-highly-capable","slug":"phi-3-technical-report-a-highly-capable","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","date":"2024-04-22","arxiv_id":"2404.14219","n_code_links":0,"syntology":null},{"paper":"/paper/cyberseceval-2-a-wide-ranging-cybersecurity","slug":"cyberseceval-2-a-wide-ranging-cybersecurity","title":"CyberSecEval 2: A Wide-Ranging Cybersecurity Evaluation Suite for Large Language Models","date":"2024-04-19","arxiv_id":"2404.13161","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/purplellama"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/sample-design-engineering-an-empirical-study","slug":"sample-design-engineering-an-empirical-study","title":"Sample Design Engineering: An Empirical Study of What Makes Good Downstream Fine-Tuning Samples for LLMs","date":"2024-04-19","arxiv_id":"2404.13033","n_code_links":1,"syntology":null},{"paper":null,"slug":"bird-a-trustworthy-bayesian-inference","title":"BIRD: A Trustworthy Bayesian Inference Framework for Large Language Models","date":"2024-04-18","arxiv_id":"2404.12494","n_code_links":0,"syntology":null},{"paper":null,"slug":"evit-event-oriented-instruction-tuning-for","title":"EVIT: Event-Oriented Instruction Tuning for Event Reasoning","date":"2024-04-18","arxiv_id":"2404.11978","n_code_links":0,"syntology":null},{"paper":"/paper/transformer-tricks-removing-weights-for","slug":"transformer-tricks-removing-weights-for","title":"Transformer tricks: Removing weights for skipless transformers","date":"2024-04-18","arxiv_id":"2404.12362","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":6,"n_instrument":3,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["openmachine-ai/transformer-tricks"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-confidence-expression-in-large","title":"Enhancing Confidence Expression in Large Language Models Through Learning from Past Experience","date":"2024-04-16","arxiv_id":"2404.10315","n_code_links":0,"syntology":null},{"paper":"/paper/hlat-high-quality-large-language-model-pre","slug":"hlat-high-quality-large-language-model-pre","title":"HLAT: High-quality Large Language Model Pre-trained on AWS Trainium","date":"2024-04-16","arxiv_id":"2404.10630","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-and-linguistic","title":"Large language models and linguistic intentionality","date":"2024-04-15","arxiv_id":"2404.09576","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-evaluators-recognize-and-favor-their-own","title":"LLM Evaluators Recognize and Favor Their Own Generations","date":"2024-04-15","arxiv_id":"2404.13076","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-fault-detection-for-large-language","title":"Evaluation and Improvement of Fault Detection for Large Language Models","date":"2024-04-14","arxiv_id":"2404.14419","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-next-token-prediction-sufficient-for-gpt","title":"Is Next Token Prediction Sufficient for GPT? Exploration on Code Logic Comprehension","date":"2024-04-13","arxiv_id":"2404.08885","n_code_links":0,"syntology":null},{"paper":"/paper/automatic-generation-and-evaluation-of","slug":"automatic-generation-and-evaluation-of","title":"Automatic Generation and Evaluation of Reading Comprehension Test Items with Large Language Models","date":"2024-04-11","arxiv_id":"2404.07720","n_code_links":2,"syntology":null},{"paper":"/paper/hgrn2-gated-linear-rnns-with-state-expansion","slug":"hgrn2-gated-linear-rnns-with-state-expansion","title":"HGRN2: Gated Linear RNNs with State Expansion","date":"2024-04-11","arxiv_id":"2404.07904","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sustcsonglin/flash-linear-attention","opennlplab/hgrn2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/adapting-llama-decoder-to-vision-transformer","slug":"adapting-llama-decoder-to-vision-transformer","title":"Adapting LLaMA Decoder to Vision Transformer","date":"2024-04-10","arxiv_id":"2404.06773","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":11,"n_instrument":1,"unverified":1,"pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 2 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["techmonsterwang/illama"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/analyzing-the-performance-of-large-language","slug":"analyzing-the-performance-of-large-language","title":"Analyzing the Performance of Large Language Models on Code Summarization","date":"2024-04-10","arxiv_id":"2404.08018","n_code_links":1,"syntology":null},{"paper":"/paper/cqil-inference-latency-optimization-with","slug":"cqil-inference-latency-optimization-with","title":"CQIL: Inference Latency Optimization with Concurrent Computation of Quasi-Independent Layers","date":"2024-04-10","arxiv_id":"2404.06709","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-concept-depth-how-large-language","slug":"exploring-concept-depth-how-large-language","title":"Exploring Concept Depth: How Large Language Models Acquire Knowledge at Different Layers?","date":"2024-04-10","arxiv_id":"2404.07066","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luckfort/cd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lossless-acceleration-of-large-language-model","slug":"lossless-acceleration-of-large-language-model","title":"Lossless Acceleration of Large Language Model via Adaptive N-gram Parallel Decoding","date":"2024-04-10","arxiv_id":"2404.08698","n_code_links":1,"syntology":null},{"paper":null,"slug":"llms-reading-comprehension-is-affected-by","title":"LLMs' Reading Comprehension Is Affected by Parametric Knowledge and Struggles with Hypothetical Statements","date":"2024-04-09","arxiv_id":"2404.06283","n_code_links":0,"syntology":null},{"paper":null,"slug":"sambalingo-teaching-large-language-models-new","title":"SambaLingo: Teaching Large Language Models New Languages","date":"2024-04-08","arxiv_id":"2404.05829","n_code_links":0,"syntology":null},{"paper":"/paper/xiwu-a-basis-flexible-and-learnable-llm-for","slug":"xiwu-a-basis-flexible-and-learnable-llm-for","title":"Xiwu: A Basis Flexible and Learnable LLM for High Energy Physics","date":"2024-04-08","arxiv_id":"2404.08001","n_code_links":1,"syntology":null},{"paper":null,"slug":"increased-llm-vulnerabilities-from-fine","title":"Fine-Tuning, Quantization, and LLMs: Navigating Unintended Outcomes","date":"2024-04-05","arxiv_id":"2404.04392","n_code_links":0,"syntology":null},{"paper":"/paper/scope-ambiguities-in-large-language-models","slug":"scope-ambiguities-in-large-language-models","title":"Scope Ambiguities in Large Language Models","date":"2024-04-05","arxiv_id":"2404.04332","n_code_links":1,"syntology":null},{"paper":"/paper/simple-techniques-for-enhancing-sentence","slug":"simple-techniques-for-enhancing-sentence","title":"Simple Techniques for Enhancing Sentence Embeddings in Generative Language Models","date":"2024-04-05","arxiv_id":"2404.03921","n_code_links":2,"syntology":null},{"paper":"/paper/teaching-llama-a-new-language-through-cross","slug":"teaching-llama-a-new-language-through-cross","title":"Teaching Llama a New Language Through Cross-Lingual Knowledge Transfer","date":"2024-04-05","arxiv_id":"2404.04042","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tartunlp/llammas"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-llms-at-detecting-errors-in-llm","slug":"evaluating-llms-at-detecting-errors-in-llm","title":"Evaluating LLMs at Detecting Errors in LLM Responses","date":"2024-04-04","arxiv_id":"2404.03602","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["psunlpgroup/realmistake"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-incomplete-loop-deductive-inductive-and","title":"An Incomplete Loop: Deductive, Inductive, and Abductive Learning in Large Language Models","date":"2024-04-03","arxiv_id":"2404.03028","n_code_links":0,"syntology":null},{"paper":"/paper/badam-a-memory-efficient-full-parameter","slug":"badam-a-memory-efficient-full-parameter","title":"BAdam: A Memory Efficient Full Parameter Optimization Method for Large Language Models","date":"2024-04-03","arxiv_id":"2404.02827","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":4,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ledzy/badam"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mimir-a-streamlined-platform-for-personalized","title":"MIMIR: A Streamlined Platform for Personalized Agent Tuning in Domain Expertise","date":"2024-04-03","arxiv_id":"2404.04285","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-inference-efficiency-of-large","title":"Enhancing Inference Efficiency of Large Language Models: Investigating Optimization Strategies and Architectural Innovations","date":"2024-04-02","arxiv_id":"2404.05741","n_code_links":0,"syntology":null},{"paper":"/paper/prego-online-mistake-detection-in-procedural","slug":"prego-online-mistake-detection-in-procedural","title":"PREGO: online mistake detection in PRocedural EGOcentric videos","date":"2024-04-02","arxiv_id":"2404.01933","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":8,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aleflabo/prego"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bailong-bilingual-transfer-learning-based-on","title":"Bailong: Bilingual Transfer Learning based on QLoRA and Zip-tie Embedding","date":"2024-04-01","arxiv_id":"2404.00862","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-prompted-mixture-of-experts-for","slug":"prompt-prompted-mixture-of-experts-for","title":"Prompt-prompted Adaptive Structured Pruning for Efficient LLM Generation","date":"2024-04-01","arxiv_id":"2404.01365","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":7,"n_instrument":2,"unverified":1,"pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hdong920/griffin"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/query-performance-prediction-using-relevance","slug":"query-performance-prediction-using-relevance","title":"Query Performance Prediction using Relevance Judgments Generated by Large Language Models","date":"2024-04-01","arxiv_id":"2404.01012","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chuanmeng/qpp-genre"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/linguistic-calibration-of-language-models","slug":"linguistic-calibration-of-language-models","title":"Linguistic Calibration of Long-Form Generations","date":"2024-03-30","arxiv_id":"2404.00474","n_code_links":1,"syntology":null},{"paper":null,"slug":"secret-keepers-the-impact-of-llms-on","title":"Secret Keepers: The Impact of LLMs on Linguistic Markers of Personal Traits","date":"2024-03-30","arxiv_id":"2404.00267","n_code_links":0,"syntology":null},{"paper":"/paper/latxa-an-open-language-model-and-evaluation","slug":"latxa-an-open-language-model-and-evaluation","title":"Latxa: An Open Language Model and Evaluation Suite for Basque","date":"2024-03-29","arxiv_id":"2403.20266","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-invalsi-benchmark-measuring-language","title":"The Invalsi Benchmarks: measuring Linguistic and Mathematical understanding of Large Language Models in Italian","date":"2024-03-27","arxiv_id":"2403.18697","n_code_links":0,"syntology":null},{"paper":null,"slug":"alisa-accelerating-large-language-model","title":"ALISA: Accelerating Large Language Model Inference via Sparsity-Aware KV Caching","date":"2024-03-26","arxiv_id":"2403.17312","n_code_links":0,"syntology":null},{"paper":"/paper/constructions-are-so-difficult-that-even","slug":"constructions-are-so-difficult-that-even","title":"Constructions Are So Difficult That Even Large Language Models Get Them Right for the Wrong Reasons","date":"2024-03-26","arxiv_id":"2403.17760","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-legal-document-retrieval-a-multi","title":"Enhancing Legal Document Retrieval: A Multi-Phase Approach with Large Language Models","date":"2024-03-26","arxiv_id":"2403.18093","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-solution-for-the-iccv-2023-1st-scientific","title":"The Solution for the ICCV 2023 1st Scientific Figure Captioning Challenge","date":"2024-03-26","arxiv_id":"2403.17342","n_code_links":0,"syntology":null},{"paper":null,"slug":"verbing-weirds-language-models-evaluation-of","title":"Verbing Weirds Language (Models): Evaluation of English Zero-Derivation in Five LLMs","date":"2024-03-26","arxiv_id":"2403.17856","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-refinement-of-project-level-code","slug":"iterative-refinement-of-project-level-code","title":"Iterative Refinement of Project-Level Code Context for Precise Code Generation with Compiler Feedback","date":"2024-03-25","arxiv_id":"2403.16792","n_code_links":1,"syntology":null},{"paper":null,"slug":"qibo-a-large-language-model-for-traditional","title":"Qibo: A Large Language Model for Traditional Chinese Medicine","date":"2024-03-24","arxiv_id":"2403.16056","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-llm-for-enterprise-practical","title":"Fine Tuning LLM for Enterprise: Practical Guidelines and Recommendations","date":"2024-03-23","arxiv_id":"2404.10779","n_code_links":0,"syntology":null},{"paper":"/paper/ghost-sentence-a-tool-for-everyday-users-to","slug":"ghost-sentence-a-tool-for-everyday-users-to","title":"Protecting Copyrighted Material with Unique Identifiers in Large Language Model Training","date":"2024-03-23","arxiv_id":"2403.15740","n_code_links":1,"syntology":null},{"paper":"/paper/llambert-large-scale-low-cost-data-annotation","slug":"llambert-large-scale-low-cost-data-annotation","title":"LlamBERT: Large-scale low-cost data annotation in NLP","date":"2024-03-23","arxiv_id":"2403.15938","n_code_links":1,"syntology":null}],"record_sha256":"d51251fb243032249a6f0865e6be24f2874adebf055fbf3e3460f9331305267f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}