{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mmlu/papers/3","list_of":"/task/mmlu","task":"MMLU","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":340,"counts":{"archive_papers_tagged":340,"with_a_code_link":156,"where_syntology_ran_a_sample":78,"not_listed_spam_title":0,"listed":340,"listed_where_code_ran":78,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":60,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":60,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mmlu","prev":"/task/mmlu/papers/2","next":"/task/mmlu/papers/4","papers":[{"url":null,"slug":"efficient-model-development-through-fine","title":"Efficient Model Development through Fine-tuning Transfer","date":"2025-03-25","arxiv_id":"2503.20110","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-bias-in-retrieval-augmented","title":"Bias Evaluation and Mitigation in Retrieval-Augmented Medical Question-Answering Systems","date":"2025-03-19","arxiv_id":"2503.15454","repositories_listed":0,"syntology":null},{"url":null,"slug":"superbpe-space-travel-for-language-models","title":"SuperBPE: Space Travel for Language Models","date":"2025-03-17","arxiv_id":"2503.13423","repositories_listed":0,"syntology":null},{"url":null,"slug":"lag-mmlu-benchmarking-frontier-llm","title":"LAG-MMLU: Benchmarking Frontier LLM Understanding in Latvian and Giriama","date":"2025-03-14","arxiv_id":"2503.11911","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmlu-prox-a-multilingual-benchmark-for","title":"MMLU-ProX: A Multilingual Benchmark for Advanced Large Language Model Evaluation","date":"2025-03-13","arxiv_id":"2503.10497","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-the-mathematical-reasoning-in","title":"Evaluating Mathematical Reasoning Across Large Language Models: A Fine-Grained Approach","date":"2025-03-13","arxiv_id":"2503.10573","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectiveness-of-zero-shot-cot-in-japanese","title":"Effectiveness of Zero-shot-CoT in Japanese Prompts","date":"2025-03-09","arxiv_id":"2503.06765","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-approximate-caching-for-faster","title":"Leveraging Approximate Caching for Faster Retrieval-Augmented Generation","date":"2025-03-07","arxiv_id":"2503.05530","repositories_listed":0,"syntology":null},{"url":null,"slug":"statistical-guarantees-of-correctness","title":"Correctness Coverage Evaluation for Medical Multiple-Choice Question Answering Based on the Enhanced Conformal Prediction Framework","date":"2025-03-07","arxiv_id":"2503.05505","repositories_listed":0,"syntology":null},{"url":null,"slug":"symbolic-mixture-of-experts-adaptive-skill","title":"Symbolic Mixture-of-Experts: Adaptive Skill-based Routing for Heterogeneous Reasoning","date":"2025-03-07","arxiv_id":"2503.05641","repositories_listed":0,"syntology":null},{"url":null,"slug":"universality-of-layer-level-entropy-weighted","title":"Universality of Layer-Level Entropy-Weighted Quantization Beyond Model Architecture and Size","date":"2025-03-06","arxiv_id":"2503.04704","repositories_listed":0,"syntology":null},{"url":null,"slug":"kurtail-kurtosis-based-llm-quantization","title":"KurTail : Kurtosis-based LLM Quantization","date":"2025-03-03","arxiv_id":"2503.01483","repositories_listed":0,"syntology":null},{"url":null,"slug":"none-of-the-above-less-of-the-right-parallel","title":"None of the Above, Less of the Right: Parallel Patterns between Humans and LLMs on Multi-Choice Questions Answering","date":"2025-03-03","arxiv_id":"2503.01550","repositories_listed":0,"syntology":null},{"url":null,"slug":"polyprompt-automating-knowledge-extraction","title":"PolyPrompt: Automating Knowledge Extraction from Multilingual Language Models with Dynamic Prompt Generation","date":"2025-02-27","arxiv_id":"2502.19756","repositories_listed":0,"syntology":null},{"url":null,"slug":"distill-not-only-data-but-also-rewards-can","title":"Distill Not Only Data but Also Rewards: Can Smaller Language Models Surpass Larger Ones?","date":"2025-02-26","arxiv_id":"2502.19557","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-federated-search-for-retrieval","title":"Efficient Federated Search for Retrieval-Augmented Generation","date":"2025-02-26","arxiv_id":"2502.19280","repositories_listed":0,"syntology":null},{"url":null,"slug":"correlating-and-predicting-human-evaluations","title":"Correlating and Predicting Human Evaluations of Language Models from Natural Language Processing Benchmarks","date":"2025-02-24","arxiv_id":"2502.18339","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-benchmark-contamination-through","title":"Detecting Benchmark Contamination Through Watermarking","date":"2025-02-24","arxiv_id":"2502.17259","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-scaling-laws-for-emergent","title":"Distributional Scaling Laws for Emergent Capabilities","date":"2025-02-24","arxiv_id":"2502.17356","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-expert-contributions-in-a-moe-llm","title":"Evaluating Expert Contributions in a MoE LLM for Quiz-Based Tasks","date":"2025-02-24","arxiv_id":"2502.17187","repositories_listed":0,"syntology":null},{"url":null,"slug":"swallowing-the-poison-pills-insights-from","title":"Swallowing the Poison Pills: Insights from Vulnerability Disparity Among LLMs","date":"2025-02-23","arxiv_id":"2502.18518","repositories_listed":0,"syntology":null},{"url":null,"slug":"obliviate-efficient-unmemorization-for","title":"Obliviate: Efficient Unmemorization for Protecting Intellectual Property in Large Language Models","date":"2025-02-20","arxiv_id":"2502.15010","repositories_listed":0,"syntology":null},{"url":null,"slug":"triangulating-llm-progress-through-benchmarks","title":"Triangulating LLM Progress through Benchmarks, Games, and Cognitive Tests","date":"2025-02-20","arxiv_id":"2502.14359","repositories_listed":0,"syntology":null},{"url":null,"slug":"none-of-the-others-a-general-technique-to","title":"None of the Others: a General Technique to Distinguish Reasoning from Memorization in Multiple-Choice LLM Evaluation Benchmarks","date":"2025-02-18","arxiv_id":"2502.12896","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-complexity-measurement-as-a-noisy","title":"Language Complexity Measurement as a Noisy Zero-Shot Proxy for Evaluating LLM Performance","date":"2025-02-17","arxiv_id":"2502.11578","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-fully-exploiting-llm-internal-states","title":"Towards Fully Exploiting LLM Internal States to Enhance Knowledge Boundary Perception","date":"2025-02-17","arxiv_id":"2502.11677","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-uncertainty-estimation-for","title":"Leveraging Uncertainty Estimation for Efficient LLM Routing","date":"2025-02-16","arxiv_id":"2502.11021","repositories_listed":0,"syntology":null},{"url":null,"slug":"octotools-an-agentic-framework-with","title":"OctoTools: An Agentic Framework with Extensible Tools for Complex Reasoning","date":"2025-02-16","arxiv_id":"2502.11271","repositories_listed":0,"syntology":null},{"url":null,"slug":"ori-o-routing-intelligence","title":"ORI: O Routing Intelligence","date":"2025-02-14","arxiv_id":"2502.10051","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-saving-llm-cascades-with-early","title":"Cost-Saving LLM Cascades with Early Abstention","date":"2025-02-13","arxiv_id":"2502.09054","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-self-to-supervised-fine-tuning-for","title":"Selective Self-to-Supervised Fine-Tuning for Generalization in Large Language Models","date":"2025-02-12","arxiv_id":"2502.08130","repositories_listed":0,"syntology":null},{"url":null,"slug":"tokenization-standards-for-linguistic","title":"Tokenization Standards for Linguistic Integrity: Turkish as a Benchmark","date":"2025-02-10","arxiv_id":"2502.07057","repositories_listed":0,"syntology":null},{"url":null,"slug":"frames-boosting-llms-with-a-four-quadrant","title":"FRAMES: Boosting LLMs with A Four-Quadrant Multi-Stage Pretraining Strategy","date":"2025-02-08","arxiv_id":"2502.05551","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapt-pruner-adaptive-structural-pruning-for","title":"Adapt-Pruner: Adaptive Structural Pruning for Efficient Small Language Model Training","date":"2025-02-05","arxiv_id":"2502.03460","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-mixture-of-agents-is-mixing","title":"Rethinking Mixture-of-Agents: Is Mixing Different Large Language Models Beneficial?","date":"2025-02-02","arxiv_id":"2502.00674","repositories_listed":0,"syntology":null},{"url":null,"slug":"indicmmlu-pro-benchmarking-the-indic-large","title":"IndicMMLU-Pro: Benchmarking Indic Large Language Models on Multi-Task Language Understanding","date":"2025-01-27","arxiv_id":"2501.15747","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardml-a-benchmark-for-evaluating-data","title":"HardML: A Benchmark For Evaluating Data Science And Machine Learning knowledge and reasoning in AI","date":"2025-01-26","arxiv_id":"2501.15627","repositories_listed":0,"syntology":null},{"url":null,"slug":"humanity-s-last-exam","title":"Humanity's Last Exam","date":"2025-01-24","arxiv_id":"2501.14249","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-reasoning-capacity-of-ai-models-and","title":"On the Reasoning Capacity of AI Models and How to Quantify It","date":"2025-01-23","arxiv_id":"2501.13833","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-your-llm-trapped-in-a-mental-set","title":"Is your LLM trapped in a Mental Set? Investigative study on how mental sets affect the reasoning capabilities of LLMs","date":"2025-01-21","arxiv_id":"2501.11833","repositories_listed":0,"syntology":null},{"url":null,"slug":"dna-1-0-technical-report","title":"DNA 1.0 Technical Report","date":"2025-01-18","arxiv_id":"2501.10648","repositories_listed":0,"syntology":null},{"url":null,"slug":"inference-time-compute-more-faithful-a","title":"Inference-Time-Compute: More Faithful? A Research Note","date":"2025-01-14","arxiv_id":"2501.08156","repositories_listed":0,"syntology":null},{"url":null,"slug":"unraveling-indirect-in-context-learning-using","title":"Unraveling Indirect In-Context Learning Using Influence Functions","date":"2025-01-01","arxiv_id":"2501.01473","repositories_listed":0,"syntology":null},{"url":null,"slug":"monty-hall-and-optimized-conformal-prediction","title":"Monty Hall and Optimized Conformal Prediction to Improve Decision-Making with LLMs","date":"2024-12-31","arxiv_id":"2501.00555","repositories_listed":0,"syntology":null},{"url":null,"slug":"setting-standards-in-turkish-nlp-tr-mmlu-for","title":"Setting Standards in Turkish NLP: TR-MMLU for Large Language Model Evaluation","date":"2024-12-31","arxiv_id":"2501.00593","repositories_listed":0,"syntology":null},{"url":null,"slug":"secbench-a-comprehensive-multi-dimensional","title":"SecBench: A Comprehensive Multi-Dimensional Benchmarking Dataset for LLMs in Cybersecurity","date":"2024-12-30","arxiv_id":"2412.20787","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixllm-llm-quantization-with-global-mixed","title":"MixLLM: LLM Quantization with Global Mixed-precision between Output-features and Highly-efficient System Design","date":"2024-12-19","arxiv_id":"2412.14590","repositories_listed":0,"syntology":null},{"url":null,"slug":"chainrank-dpo-chain-rank-direct-preference","title":"ChainRank-DPO: Chain Rank Direct Preference Optimization for LLM Rankers","date":"2024-12-18","arxiv_id":"2412.14405","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-the-secret-recipe-a-guide-for","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","date":"2024-12-17","arxiv_id":"2412.13337","repositories_listed":0,"syntology":null},{"url":null,"slug":"nanoscaling-floating-point-nxfp-nanomantissa","title":"Nanoscaling Floating-Point (NxFP): NanoMantissa, Adaptive Microexponents, and Code Recycling for Direct-Cast Compression of Large Language Models","date":"2024-12-15","arxiv_id":"2412.19821","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-distillation-for-efficient-few-shot","title":"LLM Distillation for Efficient Few-Shot Multiple Choice Question Answering","date":"2024-12-13","arxiv_id":"2412.09807","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-mmlu-understanding-and-addressing","title":"Global MMLU: Understanding and Addressing Cultural and Linguistic Biases in Multilingual Evaluation","date":"2024-12-04","arxiv_id":"2412.03304","repositories_listed":0,"syntology":null},{"url":null,"slug":"nemotron-cc-transforming-common-crawl-into-a","title":"Nemotron-CC: Transforming Common Crawl into a Refined Long-Horizon Pretraining Dataset","date":"2024-12-03","arxiv_id":"2412.02595","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-vulnerability-of-language-model","title":"The Vulnerability of Language Model Benchmarks: Do They Accurately Reflect True LLM Performance?","date":"2024-12-02","arxiv_id":"2412.03597","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-physics-reasoning-in-large-language","title":"Improving Physics Reasoning in Large Language Models Using Mixture of Refinement Agents","date":"2024-12-01","arxiv_id":"2412.00821","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-and-provable-scaling-law-for-the","title":"Simple and Provable Scaling Laws for the Test-Time Compute of Large Language Models","date":"2024-11-29","arxiv_id":"2411.19477","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-cache-conditional-experts-for","title":"Mixture of Cache-Conditional Experts for Efficient Mobile Device Inference","date":"2024-11-27","arxiv_id":"2412.00099","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-emergent-capabilities-by","title":"Predicting Emergent Capabilities by Finetuning","date":"2024-11-25","arxiv_id":"2411.16035","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentionbreaker-adaptive-evolutionary","title":"GenBFA: An Evolutionary Optimization Approach to Bit-Flip Attacks on LLMs","date":"2024-11-21","arxiv_id":"2411.13757","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-silly-questions-improves-large","title":"Learning from \"Silly\" Questions Improves Large Language Models, But Only Slightly","date":"2024-11-21","arxiv_id":"2411.14121","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-adapting-routing-rar-improving","title":"Real-time Adapting Routing (RAR): Improving Efficiency Through Continuous Learning in Software Powered by Layered Foundation Models","date":"2024-11-14","arxiv_id":"2411.09837","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-robustness-of-llms-to-adversarial","title":"Reasoning Robustness of LLMs to Adversarial Typographical Errors","date":"2024-11-08","arxiv_id":"2411.05345","repositories_listed":0,"syntology":null},{"url":null,"slug":"watson-a-cognitive-observability-framework","title":"Watson: A Cognitive Observability Framework for the Reasoning of LLM-Powered Agents","date":"2024-11-05","arxiv_id":"2411.03455","repositories_listed":0,"syntology":null},{"url":null,"slug":"project-mpg-towards-a-generalized-performance","title":"Project MPG: towards a generalized performance benchmark for LLM capabilities","date":"2024-10-28","arxiv_id":"2410.22368","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-dense-reward-understanding-the-gap","title":"Adaptive Dense Reward: Understanding the Gap Between Action and Reward Space in Alignment","date":"2024-10-23","arxiv_id":"2411.00809","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-guided-reasoning-improving-mathematical","title":"Step Guided Reasoning: Improving Mathematical Reasoning using Guidance Generation and Step Reasoning","date":"2024-10-18","arxiv_id":"2410.19817","repositories_listed":0,"syntology":null},{"url":null,"slug":"g-designer-architecting-multi-agent","title":"G-Designer: Architecting Multi-agent Communication Topologies via Graph Neural Networks","date":"2024-10-15","arxiv_id":"2410.11782","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-math-informed-synthetic-dialogues-for","title":"MIND: Math Informed syNthetic Dialogues for Pretraining LLMs","date":"2024-10-15","arxiv_id":"2410.12881","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-cross-lingual-llm-evaluation-for","title":"Towards Multilingual LLM Evaluation for European Languages","date":"2024-10-11","arxiv_id":"2410.08928","repositories_listed":0,"syntology":null},{"url":null,"slug":"upcycling-large-language-models-into-mixture","title":"Upcycling Large Language Models into Mixture of Experts","date":"2024-10-10","arxiv_id":"2410.07524","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-compression-with-neural-architecture","title":"Large Language Model Compression with Neural Architecture Search","date":"2024-10-09","arxiv_id":"2410.06479","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-paths-optimization-learning-to","title":"Reasoning Paths Optimization: Learning to Reason and Explore From Diverse Paths","date":"2024-10-07","arxiv_id":"2410.10858","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-approximations-for-improving","title":"Continuous Approximations for Improving Quantization Aware Training of LLMs","date":"2024-10-06","arxiv_id":"2410.10849","repositories_listed":0,"syntology":null},{"url":null,"slug":"braintransformers-snn-llm","title":"BrainTransformers: SNN-LLM","date":"2024-10-03","arxiv_id":"2410.14687","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiently-deploying-llms-with-controlled","title":"Efficiently Deploying LLMs with Controlled Risk","date":"2024-10-03","arxiv_id":"2410.02173","repositories_listed":0,"syntology":null},{"url":null,"slug":"dopamine-domain-specific-pre-training","title":"DoPAMine: Domain-specific Pre-training Adaptation from seed-guided data Mining","date":"2024-09-30","arxiv_id":"2410.00260","repositories_listed":0,"syntology":null},{"url":null,"slug":"instance-adaptive-zero-shot-chain-of-thought","title":"Instance-adaptive Zero-shot Chain-of-Thought Prompting","date":"2024-09-30","arxiv_id":"2409.20441","repositories_listed":0,"syntology":null},{"url":null,"slug":"progress-report-towards-european-llms","title":"Teuken-7B-Base & Teuken-7B-Instruct: Towards European LLMs","date":"2024-09-30","arxiv_id":"2410.03730","repositories_listed":0,"syntology":null},{"url":null,"slug":"ssr-alignment-aware-modality-connector-for","title":"SSR: Alignment-Aware Modality Connector for Speech Language Models","date":"2024-09-30","arxiv_id":"2410.00168","repositories_listed":0,"syntology":null},{"url":null,"slug":"2409-14026","title":"Uncovering Latent Chain of Thought Vectors in Language Models","date":"2024-09-21","arxiv_id":"2409.14026","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilingual-evaluation-of-language-models-on","title":"Bilingual Evaluation of Language Models on General Knowledge in University Entrance Exams with Minimal Contamination","date":"2024-09-19","arxiv_id":"2409.12746","repositories_listed":0,"syntology":null},{"url":null,"slug":"grin-gradient-informed-moe","title":"GRIN: GRadient-INformed MoE","date":"2024-09-18","arxiv_id":"2409.12136","repositories_listed":0,"syntology":null},{"url":null,"slug":"cpl-critical-planning-step-learning-boosts","title":"CPL: Critical Plan Step Learning Boosts LLM Generalization in Reasoning Tasks","date":"2024-09-13","arxiv_id":"2409.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"eir-thai-medical-large-language-models","title":"Eir: Thai Medical Large Language Models","date":"2024-09-13","arxiv_id":"2409.08523","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-self-rehearsal-a-fine-tuning","title":"Selective Self-Rehearsal: A Fine-Tuning Approach to Improve Generalization in Large Language Models","date":"2024-09-07","arxiv_id":"2409.04787","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-beyond-bias-a-study-on","title":"Reasoning Beyond Bias: A Study on Counterfactual Prompting and Chain of Thought Reasoning","date":"2024-08-16","arxiv_id":"2408.08651","repositories_listed":0,"syntology":null},{"url":null,"slug":"selectllm-query-aware-efficient-selection","title":"SelectLLM: Query-Aware Efficient Selection Algorithm for Large Language Models","date":"2024-08-16","arxiv_id":"2408.08545","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-02239","title":"BOTS-LM: Training Large Language Models for Setswana","date":"2024-08-05","arxiv_id":"2408.02239","repositories_listed":0,"syntology":null},{"url":null,"slug":"networks-of-networks-complexity-class","title":"Networks of Networks: Complexity Class Principles Applied to Compound AI Systems Design","date":"2024-07-23","arxiv_id":"2407.16831","repositories_listed":0,"syntology":null},{"url":null,"slug":"allam-large-language-models-for-arabic-and","title":"ALLaM: Large Language Models for Arabic and English","date":"2024-07-22","arxiv_id":"2407.15390","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentinstruct-toward-generative-teaching-with","title":"AgentInstruct: Toward Generative Teaching with Agentic Flows","date":"2024-07-03","arxiv_id":"2407.03502","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-effective-proxy-reward-model","title":"Cost-Effective Proxy Reward Model Construction with On-Policy and Active Learning","date":"2024-07-02","arxiv_id":"2407.02119","repositories_listed":0,"syntology":null},{"url":null,"slug":"changing-answer-order-can-decrease-mmlu","title":"Changing Answer Order Can Decrease MMLU Accuracy","date":"2024-06-27","arxiv_id":"2406.19470","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-evaluation-of-large-language","title":"Data Efficient Evaluation of Large Language Models and Text-to-Image Models via Adaptive Sampling","date":"2024-06-21","arxiv_id":"2406.15527","repositories_listed":0,"syntology":null},{"url":null,"slug":"dem-distribution-edited-model-for-training","title":"DEM: Distribution Edited Model for Training with Mixed Data Distributions","date":"2024-06-21","arxiv_id":"2406.15570","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimised-grouped-query-attention-mechanism","title":"Optimised Grouped-Query Attention Mechanism for Transformers","date":"2024-06-21","arxiv_id":"2406.14963","repositories_listed":0,"syntology":null},{"url":null,"slug":"pistis-rag-a-scalable-cascading-framework","title":"Pistis-RAG: Enhancing Retrieval-Augmented Generation with Human Feedback","date":"2024-06-21","arxiv_id":"2407.00072","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-finetuning-for-factual-1","title":"Understanding Finetuning for Factual Knowledge Extraction","date":"2024-06-20","arxiv_id":"2406.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"cultural-conditioning-or-placebo-on-the","title":"Cultural Conditioning or Placebo? On the Effectiveness of Socio-Demographic Prompting","date":"2024-06-17","arxiv_id":"2406.11661","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-base-rate-effect-on-llm-benchmark","title":"The Base-Rate Effect on LLM Benchmark Performance: Disambiguating Test-Taking Strategies from Benchmark Performance","date":"2024-06-17","arxiv_id":"2406.11634","repositories_listed":0,"syntology":null}],"record_sha256":"bd7c802991bac373e356c96f11261697c65c288a77e28971e793c8f90123c41a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}