{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/llama/papers/9","list_of":"/method/llama","method":"LLaMA","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":11,"rows_per_page":100,"rows":[801,900],"of":1062,"counts":{"archive_papers_tagged":1062,"with_a_code_link":423,"where_syntology_ran_a_sample":143,"not_listed_spam_title":0,"listed":1062,"listed_where_code_ran":143,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":127,"every_run_a_failure_of_syntologys_instrument":16,"listed_with_a_run_with_no_instrument_failure":127,"listed_every_run_a_failure_of_syntologys_instrument":16,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/llama","prev":"/method/llama/papers/8","next":"/method/llama/papers/10","papers":[{"paper":null,"slug":"actionable-cyber-threat-intelligence-using","title":"Actionable Cyber Threat Intelligence using Knowledge Graphs and Large Language Models","date":"2024-06-30","arxiv_id":"2407.02528","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-power-scheduling-a","slug":"large-language-models-for-power-scheduling-a","title":"Large Language Models for Power Scheduling: A User-Centric Approach","date":"2024-06-29","arxiv_id":"2407.00476","n_code_links":1,"syntology":null},{"paper":"/paper/mixture-of-in-context-experts-enhance-llms","slug":"mixture-of-in-context-experts-enhance-llms","title":"Mixture of In-Context Experts Enhance LLMs' Long Context Awareness","date":"2024-06-28","arxiv_id":"2406.19598","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["p1nksnow/moice"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/paraphrase-types-elicit-prompt-engineering","slug":"paraphrase-types-elicit-prompt-engineering","title":"Paraphrase Types Elicit Prompt Engineering Capabilities","date":"2024-06-28","arxiv_id":"2406.19898","n_code_links":2,"syntology":null},{"paper":"/paper/understanding-and-mitigating-language","slug":"understanding-and-mitigating-language","title":"Understanding and Mitigating Language Confusion in LLMs","date":"2024-06-28","arxiv_id":"2406.20052","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["for-ai/language-confusion"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/arzen-llm-code-switched-egyptian-arabic","slug":"arzen-llm-code-switched-egyptian-arabic","title":"ArzEn-LLM: Code-Switched Egyptian Arabic-English Translation and Speech Recognition Using LLMs","date":"2024-06-26","arxiv_id":"2406.18120","n_code_links":1,"syntology":null},{"paper":"/paper/blockllm-memory-efficient-adaptation-of-llms","slug":"blockllm-memory-efficient-adaptation-of-llms","title":"BlockLLM: Memory-Efficient Adaptation of LLMs by Selecting and Optimizing the Right Coordinate Blocks","date":"2024-06-25","arxiv_id":"2406.17296","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["RAmruthaVignesh/blockllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"following-length-constraints-in-instructions","title":"Following Length Constraints in Instructions","date":"2024-06-25","arxiv_id":"2406.17744","n_code_links":0,"syntology":null},{"paper":"/paper/grass-compute-efficient-low-memory-llm","slug":"grass-compute-efficient-low-memory-llm","title":"Grass: Compute Efficient Low-Memory LLM Training with Structured Sparse Gradients","date":"2024-06-25","arxiv_id":"2406.17660","n_code_links":1,"syntology":null},{"paper":"/paper/t-mac-cpu-renaissance-via-table-lookup-for","slug":"t-mac-cpu-renaissance-via-table-lookup-for","title":"T-MAC: CPU Renaissance via Table Lookup for Low-Bit LLM Deployment on Edge","date":"2024-06-25","arxiv_id":"2407.00088","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["microsoft/t-mac"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-fineweb-datasets-decanting-the-web-for","slug":"the-fineweb-datasets-decanting-the-web-for","title":"The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale","date":"2024-06-25","arxiv_id":"2406.17557","n_code_links":1,"syntology":null},{"paper":null,"slug":"unmasking-the-imposters-in-domain-detection","title":"Unmasking the Imposters: How Censorship and Domain Adaptation Affect the Detection of Machine-Generated Tweets","date":"2024-06-25","arxiv_id":"2406.17967","n_code_links":0,"syntology":null},{"paper":"/paper/adam-mini-use-fewer-learning-rates-to-gain","slug":"adam-mini-use-fewer-learning-rates-to-gain","title":"Adam-mini: Use Fewer Learning Rates To Gain More","date":"2024-06-24","arxiv_id":"2406.16793","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zyushun/adam-mini"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/autodetect-towards-a-unified-framework-for","slug":"autodetect-towards-a-unified-framework-for","title":"AutoDetect: Towards a Unified Framework for Automated Weakness Detection in Large Language Models","date":"2024-06-24","arxiv_id":"2406.16714","n_code_links":1,"syntology":null},{"paper":null,"slug":"paraphrase-and-aggregate-with-large-language","title":"Paraphrase and Aggregate with Large Language Models for Minimizing Intent Classification Errors","date":"2024-06-24","arxiv_id":"2406.17163","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-laws-for-linear-complexity-language","slug":"scaling-laws-for-linear-complexity-language","title":"Scaling Laws for Linear Complexity Language Models","date":"2024-06-24","arxiv_id":"2406.16690","n_code_links":1,"syntology":null},{"paper":"/paper/fastmem-fast-memorization-of-prompt-improves","slug":"fastmem-fast-memorization-of-prompt-improves","title":"FastMem: Fast Memorization of Prompt Improves Context Awareness of Large Language Models","date":"2024-06-23","arxiv_id":"2406.16069","n_code_links":1,"syntology":null},{"paper":"/paper/dabl-detecting-semantic-anomalies-in-business","slug":"dabl-detecting-semantic-anomalies-in-business","title":"DABL: Detecting Semantic Anomalies in Business Processes Using Large Language Models","date":"2024-06-22","arxiv_id":"2406.15781","n_code_links":1,"syntology":null},{"paper":"/paper/v-recs-a-low-cost-llm4vis-recommender-with","slug":"v-recs-a-low-cost-llm4vis-recommender-with","title":"V-RECS, a Low-Cost LLM4VIS Recommender with Explanations, Captioning and Suggestions","date":"2024-06-21","arxiv_id":"2406.15259","n_code_links":1,"syntology":null},{"paper":"/paper/realhf-optimized-rlhf-training-for-large","slug":"realhf-optimized-rlhf-training-for-large","title":"ReaL: Efficient RLHF Training of Large Language Models with Parameter Reallocation","date":"2024-06-20","arxiv_id":"2406.14088","n_code_links":1,"syntology":null},{"paper":"/paper/enhancing-automated-audio-captioning-via","slug":"enhancing-automated-audio-captioning-via","title":"Enhancing Automated Audio Captioning via Large Language Models with Optimized Audio Encoding","date":"2024-06-19","arxiv_id":"2406.13275","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":4,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["frankenliu/LOAE"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/morehopqa-more-than-multi-hop-reasoning","slug":"morehopqa-more-than-multi-hop-reasoning","title":"MoreHopQA: More Than Multi-hop Reasoning","date":"2024-06-19","arxiv_id":"2406.13397","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-rags-to-rich-parameters-probing-how","title":"From RAGs to rich parameters: Probing how language models utilize external knowledge over parametric information for factual queries","date":"2024-06-18","arxiv_id":"2406.12824","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpreting-bias-in-large-language-models-a","title":"Interpreting Bias in Large Language Models: A Feature-Based Approach","date":"2024-06-18","arxiv_id":"2406.12347","n_code_links":0,"syntology":null},{"paper":null,"slug":"jailbreak-paradox-the-achilles-heel-of-llms","title":"[WIP] Jailbreak Paradox: The Achilles' Heel of LLMs","date":"2024-06-18","arxiv_id":"2406.12702","n_code_links":0,"syntology":null},{"paper":"/paper/liar-liar-logical-mire-a-benchmark-for","slug":"liar-liar-logical-mire-a-benchmark-for","title":"Liar, Liar, Logical Mire: A Benchmark for Suppositional Reasoning in Large Language Models","date":"2024-06-18","arxiv_id":"2406.12546","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mainlp/TruthQuest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"moyu-a-theoretical-study-on-massive-over","title":"MOYU: A Theoretical Study on Massive Over-activation Yielded Uplifts in LLMs","date":"2024-06-18","arxiv_id":"2406.12569","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-principles-behind-opinion-dynamics-in","title":"On the Principles behind Opinion Dynamics in Multi-Agent Systems of Large Language Models","date":"2024-06-18","arxiv_id":"2406.15492","n_code_links":0,"syntology":null},{"paper":null,"slug":"qog-question-and-options-generation-based-on","title":"QOG:Question and Options Generation based on Language Model","date":"2024-06-18","arxiv_id":"2406.12381","n_code_links":0,"syntology":null},{"paper":null,"slug":"urbanllm-autonomous-urban-activity-planning","title":"UrbanLLM: Autonomous Urban Activity Planning and Management with Large Language Models","date":"2024-06-18","arxiv_id":"2406.12360","n_code_links":0,"syntology":null},{"paper":null,"slug":"you-gotta-be-a-doctor-lin-an-investigation-of","title":"\"You Gotta be a Doctor, Lin\": An Investigation of Name-Based Bias of Large Language Models in Employment Recommendations","date":"2024-06-18","arxiv_id":"2406.12232","n_code_links":0,"syntology":null},{"paper":"/paper/ai-news-content-farms-are-easy-to-make-and","slug":"ai-news-content-farms-are-easy-to-make-and","title":"AI \"News\" Content Farms Are Easy to Make and Hard to Detect: A Case Study in Italian","date":"2024-06-17","arxiv_id":"2406.12128","n_code_links":0,"syntology":{"ran":12,"of":17,"n_ran_checked":12,"n_instrument":0,"unverified":5,"pointer_only":17,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":null}},{"paper":"/paper/analysing-zero-shot-temporal-relation","slug":"analysing-zero-shot-temporal-relation","title":"Analysing zero-shot temporal relation extraction on clinical notes using temporal consistency","date":"2024-06-17","arxiv_id":"2406.11486","n_code_links":1,"syntology":null},{"paper":"/paper/datacomp-lm-in-search-of-the-next-generation","slug":"datacomp-lm-in-search-of-the-next-generation","title":"DataComp-LM: In search of the next generation of training sets for language models","date":"2024-06-17","arxiv_id":"2406.11794","n_code_links":3,"syntology":{"ran":22,"of":22,"n_ran_checked":20,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/dialogue-action-tokens-steering-language","slug":"dialogue-action-tokens-steering-language","title":"Dialogue Action Tokens: Steering Language Models in Goal-Directed Dialogue with a Multi-Turn Planner","date":"2024-06-17","arxiv_id":"2406.11978","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["likenneth/dialogue_action_token"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/emotion-llama-multimodal-emotion-recognition","slug":"emotion-llama-multimodal-emotion-recognition","title":"Emotion-LLaMA: Multimodal Emotion Recognition and Reasoning with Instruction Tuning","date":"2024-06-17","arxiv_id":"2406.11161","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["zebangcheng/emotion-llama"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"exploring-safety-utility-trade-offs-in","title":"Exploring Safety-Utility Trade-Offs in Personalized Language Models","date":"2024-06-17","arxiv_id":"2406.11107","n_code_links":0,"syntology":null},{"paper":"/paper/generative-visual-instruction-tuning","slug":"generative-visual-instruction-tuning","title":"Generative Visual Instruction Tuning","date":"2024-06-17","arxiv_id":"2406.11262","n_code_links":1,"syntology":null},{"paper":"/paper/is-poisoning-a-real-threat-to-llm-alignment","slug":"is-poisoning-a-real-threat-to-llm-alignment","title":"Is poisoning a real threat to LLM alignment? Maybe more so than you think","date":"2024-06-17","arxiv_id":"2406.12091","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["pankayaraj/RLHFPoisoning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-scale-transfer-learning-for-tabular","slug":"large-scale-transfer-learning-for-tabular","title":"Large Scale Transfer Learning for Tabular Data via Language Modeling","date":"2024-06-17","arxiv_id":"2406.12031","n_code_links":2,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mlfoundations/rtfm","mlfoundations/tabliblib"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-the-codebook-size-of-vqgan-to-100000","slug":"scaling-the-codebook-size-of-vqgan-to-100000","title":"Scaling the Codebook Size of VQGAN to 100,000 with a Utilization Rate of 99%","date":"2024-06-17","arxiv_id":"2406.11837","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zh460045050/vqgan-lc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"self-and-cross-model-distillation-for-llms","title":"Self and Cross-Model Distillation for LLMs: Effective Methods for Refusal Pattern Alignment","date":"2024-06-17","arxiv_id":"2406.11285","n_code_links":0,"syntology":null},{"paper":"/paper/analyzing-key-neurons-in-large-language","slug":"analyzing-key-neurons-in-large-language","title":"Identifying Query-Relevant Neurons in Large Language Models for Long-Form Texts","date":"2024-06-16","arxiv_id":"2406.10868","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tigerchen52/qrneuron"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":0,"n_instrument":5,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"optimization-based-structural-pruning-for","title":"Bypass Back-propagation: Optimization-based Structural Pruning for Large Language Models via Policy Gradient","date":"2024-06-15","arxiv_id":"2406.10576","n_code_links":0,"syntology":null},{"paper":null,"slug":"reactor-mk-1-performances-mmlu-humaneval-and","title":"Reactor Mk.1 performances: MMLU, HumanEval and BBH test results","date":"2024-06-15","arxiv_id":"2406.10515","n_code_links":0,"syntology":null},{"paper":"/paper/carllava-vision-language-models-for-camera","slug":"carllava-vision-language-models-for-camera","title":"CarLLaVA: Vision language models for camera-only closed-loop driving","date":"2024-06-14","arxiv_id":"2406.10165","n_code_links":1,"syntology":null},{"paper":null,"slug":"geb-1-3b-open-lightweight-large-language","title":"GEB-1.3B: Open Lightweight Large Language Model","date":"2024-06-14","arxiv_id":"2406.09900","n_code_links":0,"syntology":null},{"paper":"/paper/defan-definitive-answer-dataset-for-llms","slug":"defan-definitive-answer-dataset-for-llms","title":"DefAn: Definitive Answer Dataset for LLMs Hallucination Evaluation","date":"2024-06-13","arxiv_id":"2406.09155","n_code_links":1,"syntology":null},{"paper":"/paper/openvla-an-open-source-vision-language-action","slug":"openvla-an-open-source-vision-language-action","title":"OpenVLA: An Open-Source Vision-Language-Action Model","date":"2024-06-13","arxiv_id":"2406.09246","n_code_links":3,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"mistral-c2f-coarse-to-fine-actor-for","title":"Mistral-C2F: Coarse to Fine Actor for Analytical and Reasoning Enhancement in RLHF and Effective-Merged LLMs","date":"2024-06-12","arxiv_id":"2406.08657","n_code_links":0,"syntology":null},{"paper":null,"slug":"underneath-the-numbers-quantitative-and","title":"Underneath the Numbers: Quantitative and Qualitative Gender Fairness in LLMs for Depression Prediction","date":"2024-06-12","arxiv_id":"2406.08183","n_code_links":0,"syntology":null},{"paper":"/paper/never-miss-a-beat-an-efficient-recipe-for","slug":"never-miss-a-beat-an-efficient-recipe-for","title":"An Efficient Recipe for Long Context Extension via Middle-Focused Positional Encoding","date":"2024-06-11","arxiv_id":"2406.07138","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bigai-nlco/cream"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/when-linear-attention-meets-autoregressive","slug":"when-linear-attention-meets-autoregressive","title":"When Linear Attention Meets Autoregressive Decoding: Towards More Effective and Efficient Linearized Large Language Models","date":"2024-06-11","arxiv_id":"2406.07368","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":3,"n_instrument":4,"unverified":4,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gatech-eic/linearized-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/language-models-are-alignable-decision-makers","slug":"language-models-are-alignable-decision-makers","title":"Language Models are Alignable Decision-Makers: Dataset and Application to the Medical Triage Domain","date":"2024-06-10","arxiv_id":"2406.06435","n_code_links":1,"syntology":null},{"paper":"/paper/lawgpt-a-chinese-legal-knowledge-enhanced","slug":"lawgpt-a-chinese-legal-knowledge-enhanced","title":"LawGPT: A Chinese Legal Knowledge-Enhanced Large Language Model","date":"2024-06-07","arxiv_id":"2406.04614","n_code_links":2,"syntology":null},{"paper":null,"slug":"assessing-the-emergent-symbolic-reasoning","title":"Assessing the Emergent Symbolic Reasoning Abilities of Llama Large Language Models","date":"2024-06-05","arxiv_id":"2406.06588","n_code_links":0,"syntology":null},{"paper":null,"slug":"irokobench-a-new-benchmark-for-african","title":"IrokoBench: A New Benchmark for African Languages in the Age of Large Language Models","date":"2024-06-05","arxiv_id":"2406.03368","n_code_links":0,"syntology":null},{"paper":"/paper/pruner-zero-evolving-symbolic-pruning-metric","slug":"pruner-zero-evolving-symbolic-pruning-metric","title":"Pruner-Zero: Evolving Symbolic Pruning Metric from scratch for Large Language Models","date":"2024-06-05","arxiv_id":"2406.02924","n_code_links":1,"syntology":{"ran":8,"of":17,"n_ran_checked":5,"n_instrument":3,"unverified":9,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","official":{"repos":["pprp/pruner-zero"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"language-models-do-hard-arithmetic-tasks","title":"Language Models Do Hard Arithmetic Tasks Easily and Hardly Do Easy Arithmetic Tasks","date":"2024-06-04","arxiv_id":"2406.02356","n_code_links":0,"syntology":null},{"paper":null,"slug":"occamllm-fast-and-exact-language-model","title":"OccamLLM: Fast and Exact Language Model Arithmetic in a Single Step","date":"2024-06-04","arxiv_id":"2406.06576","n_code_links":0,"syntology":null},{"paper":"/paper/sltrain-a-sparse-plus-low-rank-approach-for","slug":"sltrain-a-sparse-plus-low-rank-approach-for","title":"SLTrain: a sparse plus low-rank approach for parameter and memory efficient pretraining","date":"2024-06-04","arxiv_id":"2406.02214","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["andyjm3/SLTrain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"annotation-guidelines-based-knowledge","title":"Annotation Guidelines-Based Knowledge Augmentation: Towards Enhancing Large Language Models for Educational Text Classification","date":"2024-06-03","arxiv_id":"2406.00954","n_code_links":0,"syntology":null},{"paper":"/paper/demystifying-platform-requirements-for","slug":"demystifying-platform-requirements-for","title":"Demystifying AI Platform Design for Distributed Inference of Next-Generation LLM models","date":"2024-06-03","arxiv_id":"2406.01698","n_code_links":1,"syntology":null},{"paper":null,"slug":"llms-beyond-english-scaling-the-multilingual","title":"LLMs Beyond English: Scaling the Multilingual Capability of LLMs with Cross-Lingual Feedback","date":"2024-06-03","arxiv_id":"2406.01771","n_code_links":0,"syntology":null},{"paper":null,"slug":"rag-enabled-conversations-about-household","title":"Natural Language Interaction with a Household Electricity Knowledge-based Digital Twin","date":"2024-06-03","arxiv_id":"2406.06566","n_code_links":0,"syntology":null},{"paper":null,"slug":"revolutionizing-large-language-model-training","title":"SwitchLoRA: Switched Low-Rank Adaptation Can Learn Full-Rank Information","date":"2024-06-03","arxiv_id":"2406.06564","n_code_links":0,"syntology":null},{"paper":"/paper/subllm-a-novel-efficient-architecture-with","slug":"subllm-a-novel-efficient-architecture-with","title":"SUBLLM: A Novel Efficient Architecture with Token Sequence Subsampling for LLM","date":"2024-06-03","arxiv_id":"2406.06571","n_code_links":1,"syntology":null},{"paper":"/paper/magr-weight-magnitude-reduction-for-enhancing","slug":"magr-weight-magnitude-reduction-for-enhancing","title":"MagR: Weight Magnitude Reduction for Enhancing Post-Training Quantization","date":"2024-06-02","arxiv_id":"2406.00800","n_code_links":1,"syntology":{"ran":8,"of":12,"n_ran_checked":1,"n_instrument":7,"unverified":4,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 4 unverified","official":{"repos":["aozhongzhang/magr"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/audiolcm-text-to-audio-generation-with-latent","slug":"audiolcm-text-to-audio-generation-with-latent","title":"AudioLCM: Text-to-Audio Generation with Latent Consistency Models","date":"2024-06-01","arxiv_id":"2406.00356","n_code_links":2,"syntology":null},{"paper":null,"slug":"effective-interplay-between-sparsity-and","title":"Effective Interplay between Sparsity and Quantization: From Theory to Practice","date":"2024-05-31","arxiv_id":"2405.20935","n_code_links":0,"syntology":null},{"paper":"/paper/improving-generalization-and-convergence-by","slug":"improving-generalization-and-convergence-by","title":"Improving Generalization and Convergence by Enhancing Implicit Regularization","date":"2024-05-31","arxiv_id":"2405.20763","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":4,"n_instrument":3,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wmz9/ire-algorithm-framework"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-a-new-approach-for","title":"Large Language Models: A New Approach for Privacy Policy Analysis at Scale","date":"2024-05-31","arxiv_id":"2405.20900","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-point-of-view-of-a-sentiment-towards","title":"The Point of View of a Sentiment: Towards Clinician Bias Detection in Psychiatric Notes","date":"2024-05-31","arxiv_id":"2405.20582","n_code_links":0,"syntology":null},{"paper":null,"slug":"context-injection-attacks-on-large-language","title":"Hidden in Plain Sight: Exploring Chat History Tampering in Interactive Language Models","date":"2024-05-30","arxiv_id":"2405.20234","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-watermark-stealing-with","title":"Large Language Model Watermark Stealing With Mixed Integer Programming","date":"2024-05-30","arxiv_id":"2405.19677","n_code_links":0,"syntology":null},{"paper":"/paper/the-fine-tuning-paradox-boosting-translation","slug":"the-fine-tuning-paradox-boosting-translation","title":"The Fine-Tuning Paradox: Boosting Translation Quality Without Sacrificing LLM Abilities","date":"2024-05-30","arxiv_id":"2405.20089","n_code_links":1,"syntology":null},{"paper":null,"slug":"genshin-general-shield-for-natural-language","title":"Genshin: General Shield for Natural Language Processing with Large Language Models","date":"2024-05-29","arxiv_id":"2405.18741","n_code_links":0,"syntology":null},{"paper":"/paper/aligning-to-thousands-of-preferences-via","slug":"aligning-to-thousands-of-preferences-via","title":"Aligning to Thousands of Preferences via System Message Generalization","date":"2024-05-28","arxiv_id":"2405.17977","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kaistAI/Janus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/on-the-origin-of-llamas-model-tree-heritage","slug":"on-the-origin-of-llamas-model-tree-heritage","title":"On the Origin of Llamas: Model Tree Heritage Recovery","date":"2024-05-28","arxiv_id":"2405.18432","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"0 ran · 2 unverified","official":{"repos":["eliahuhorwitz/MoTHer"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/online-merging-optimizers-for-boosting","slug":"online-merging-optimizers-for-boosting","title":"Online Merging Optimizers for Boosting Rewards and Mitigating Tax in Alignment","date":"2024-05-28","arxiv_id":"2405.17931","n_code_links":1,"syntology":null},{"paper":null,"slug":"proof-of-quality-a-costless-paradigm-for","title":"Proof of Quality: A Costless Paradigm for Trustless Generative AI Model Inference on Blockchains","date":"2024-05-28","arxiv_id":"2405.17934","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-intrinsic-socioeconomic-biases","title":"Understanding Intrinsic Socioeconomic Biases in Large Language Models","date":"2024-05-28","arxiv_id":"2405.18662","n_code_links":0,"syntology":null},{"paper":"/paper/velora-memory-efficient-training-using-rank-1","slug":"velora-memory-efficient-training-using-rank-1","title":"VeLoRA: Memory Efficient Training using Rank-1 Sub-Token Projections","date":"2024-05-28","arxiv_id":"2405.17991","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":0,"n_instrument":3,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["roymiles/VeLoRA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/deeperimpact-optimizing-sparse-learned-index","slug":"deeperimpact-optimizing-sparse-learned-index","title":"DeeperImpact: Optimizing Sparse Learned Index Structures","date":"2024-05-27","arxiv_id":"2405.17093","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-and-steering-the-moral-compass-of","slug":"exploring-and-steering-the-moral-compass-of","title":"Exploring and steering the moral compass of Large Language Models","date":"2024-05-27","arxiv_id":"2405.17345","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["atlaie/ethical-llms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"textit-trans-lora-towards-data-free","title":"$\\textit{Trans-LoRA}$: towards data-free Transferable Parameter Efficient Finetuning","date":"2024-05-27","arxiv_id":"2405.17258","n_code_links":0,"syntology":null},{"paper":"/paper/spp-sparsity-preserved-parameter-efficient","slug":"spp-sparsity-preserved-parameter-efficient","title":"SPP: Sparsity-Preserved Parameter-Efficient Fine-Tuning for Large Language Models","date":"2024-05-25","arxiv_id":"2405.16057","n_code_links":1,"syntology":null},{"paper":null,"slug":"basis-selection-low-rank-decomposition-of","title":"Basis Selection: Low-Rank Decomposition of Pretrained Large Language Models for Target Applications","date":"2024-05-24","arxiv_id":"2405.15877","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-pre-trained-large-language","title":"Benchmarking the Performance of Pre-trained LLMs across Urdu NLP Tasks","date":"2024-05-24","arxiv_id":"2405.15453","n_code_links":0,"syntology":null},{"paper":null,"slug":"bisup-bidirectional-quantization-error","title":"BiSup: Bidirectional Quantization Error Suppression for Large Language Models","date":"2024-05-24","arxiv_id":"2405.15346","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-the-adversarial-robustness-of-1","slug":"evaluating-the-adversarial-robustness-of-1","title":"Evaluating and Safeguarding the Adversarial Robustness of Retrieval-Based In-Context Learning","date":"2024-05-24","arxiv_id":"2405.15984","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["simonucl/adv-retreival-icl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gecko-generative-language-model-for-english","title":"GECKO: Generative Language Model for English, Code and Korean","date":"2024-05-24","arxiv_id":"2405.15640","n_code_links":0,"syntology":null},{"paper":"/paper/optimizing-large-language-models-for-openapi","slug":"optimizing-large-language-models-for-openapi","title":"Optimizing Large Language Models for OpenAPI Code Completion","date":"2024-05-24","arxiv_id":"2405.15729","n_code_links":2,"syntology":null},{"paper":"/paper/sparse-expansion-and-neuronal-disentanglement","slug":"sparse-expansion-and-neuronal-disentanglement","title":"Sparse Expansion and Neuronal Disentanglement","date":"2024-05-24","arxiv_id":"2405.15756","n_code_links":1,"syntology":{"ran":4,"of":9,"n_ran_checked":3,"n_instrument":1,"unverified":5,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["shavit-lab/sparse-expansion"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/sparse-matrix-in-large-language-model-fine","slug":"sparse-matrix-in-large-language-model-fine","title":"Sparse Matrix in Large Language Model Fine-tuning","date":"2024-05-24","arxiv_id":"2405.15525","n_code_links":1,"syntology":null},{"paper":null,"slug":"base-of-rope-bounds-context-length","title":"Base of RoPE Bounds Context Length","date":"2024-05-23","arxiv_id":"2405.14591","n_code_links":0,"syntology":null},{"paper":"/paper/mitigating-quantization-errors-due-to","slug":"mitigating-quantization-errors-due-to","title":"Mitigating Quantization Errors Due to Activation Spikes in GLU-Based LLMs","date":"2024-05-23","arxiv_id":"2405.14428","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["onnoo/activation-spikes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/not-all-language-model-features-are-linear","slug":"not-all-language-model-features-are-linear","title":"Not All Language Model Features Are Linear","date":"2024-05-23","arxiv_id":"2405.14860","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["joshengels/multidimensionalfeatures"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/pv-tuning-beyond-straight-through-estimation","slug":"pv-tuning-beyond-straight-through-estimation","title":"PV-Tuning: Beyond Straight-Through Estimation for Extreme LLM Compression","date":"2024-05-23","arxiv_id":"2405.14852","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["vahe1994/aqlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}}],"record_sha256":"351692f1bd2b6edc1dfeccfdffe07c93953925b9bf3b1aede63de7eb70350a22","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}