{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/mmlu/papers/4","list_of":"/task/mmlu","task":"MMLU","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":4,"rows_per_page":100,"rows":[301,340],"of":340,"counts":{"archive_papers_tagged":340,"with_a_code_link":156,"where_syntology_ran_a_sample":78,"not_listed_spam_title":0,"listed":340,"listed_where_code_ran":78,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":60,"every_run_a_failure_of_syntologys_instrument":18,"listed_with_a_run_with_no_instrument_failure":60,"listed_every_run_a_failure_of_syntologys_instrument":18,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/mmlu","prev":"/task/mmlu/papers/3","next":null,"papers":[{"url":null,"slug":"reactor-mk-1-performances-mmlu-humaneval-and","title":"Reactor Mk.1 performances: MMLU, HumanEval and BBH test results","date":"2024-06-15","arxiv_id":"2406.10515","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-or-simply-next-token-prediction-a","title":"MMLU-SR: A Benchmark for Stress-Testing Reasoning Capability of Large Language Models","date":"2024-06-15","arxiv_id":"2406.15468","repositories_listed":0,"syntology":null},{"url":null,"slug":"geb-1-3b-open-lightweight-large-language","title":"GEB-1.3B: Open Lightweight Large Language Model","date":"2024-06-14","arxiv_id":"2406.09900","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-variance-in-evaluation-benchmarks","title":"Quantifying Variance in Evaluation Benchmarks","date":"2024-06-14","arxiv_id":"2406.10229","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-your-data-spark-joy-performance-gains","title":"Does your data spark joy? Performance gains from domain upsampling at the end of training","date":"2024-06-05","arxiv_id":"2406.03476","repositories_listed":0,"syntology":null},{"url":"/paper/mixeval-deriving-wisdom-of-the-crowd-from-llm","slug":"mixeval-deriving-wisdom-of-the-crowd-from-llm","title":"MixEval: Deriving Wisdom of the Crowd from LLM Benchmark Mixtures","date":"2024-06-03","arxiv_id":"2406.06565","repositories_listed":0,"syntology":null},{"url":null,"slug":"spanish-and-llm-benchmarks-is-mmlu-lost-in","title":"Spanish and LLM Benchmarks: is MMLU Lost in Translation?","date":"2024-05-28","arxiv_id":"2406.17789","repositories_listed":0,"syntology":null},{"url":null,"slug":"gecko-generative-language-model-for-english","title":"GECKO: Generative Language Model for English, Code and Korean","date":"2024-05-24","arxiv_id":"2405.15640","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-assessment-of-model-on-model-deception","title":"An Assessment of Model-On-Model Deception","date":"2024-05-10","arxiv_id":"2405.12999","repositories_listed":0,"syntology":null},{"url":null,"slug":"sutra-scalable-multilingual-language-model","title":"SUTRA: Scalable Multilingual Language Model Architecture","date":"2024-05-07","arxiv_id":"2405.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"octopus-v4-graph-of-language-models","title":"Octopus v4: Graph of language models","date":"2024-04-30","arxiv_id":"2404.19296","repositories_listed":0,"syntology":null},{"url":"/paper/phi-3-technical-report-a-highly-capable","slug":"phi-3-technical-report-a-highly-capable","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","date":"2024-04-22","arxiv_id":"2404.14219","repositories_listed":0,"syntology":null},{"url":null,"slug":"reka-core-flash-and-edge-a-series-of-powerful","title":"Reka Core, Flash, and Edge: A Series of Powerful Multimodal Language Models","date":"2024-04-18","arxiv_id":"2404.12387","repositories_listed":0,"syntology":null},{"url":null,"slug":"llama-excitor-general-instruction-tuning-via","title":"LLaMA-Excitor: General Instruction Tuning via Indirect Feature Interaction","date":"2024-04-01","arxiv_id":"2404.00913","repositories_listed":0,"syntology":null},{"url":null,"slug":"numerologic-number-encoding-for-enhanced-llms","title":"NumeroLogic: Number Encoding for Enhanced LLMs' Numerical Reasoning","date":"2024-03-30","arxiv_id":"2404.00459","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-recalibration-of-language-models","title":"Few-Shot Recalibration of Language Models","date":"2024-03-27","arxiv_id":"2403.18286","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-three-phases-sft-hybrid-model-integrated","title":"CodingTeachLLM: Empowering LLM's Coding Ability via AST Prior Knowledge","date":"2024-03-13","arxiv_id":"2403.15426","repositories_listed":0,"syntology":null},{"url":"/paper/the-claude-3-model-family-opus-sonnet-haiku","slug":"the-claude-3-model-family-opus-sonnet-haiku","title":"The Claude 3 Model Family: Opus, Sonnet, Haiku","date":"2024-03-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kormedmcqa-multi-choice-question-answering","title":"KorMedMCQA: Multi-Choice Question Answering Benchmark for Korean Healthcare Professional Licensing Examinations","date":"2024-03-03","arxiv_id":"2403.01469","repositories_listed":0,"syntology":null},{"url":null,"slug":"openmedlm-prompt-engineering-can-out-perform","title":"OpenMedLM: Prompt engineering can out-perform fine-tuning in medical question-answering with open-source large language models","date":"2024-02-29","arxiv_id":"2402.19371","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-large-language-models-mirror-cognitive","title":"Do Large Language Models Mirror Cognitive Language Processing?","date":"2024-02-28","arxiv_id":"2402.18023","repositories_listed":0,"syntology":null},{"url":null,"slug":"arl2-aligning-retrievers-for-black-box-large","title":"ARL2: Aligning Retrievers for Black-box Large Language Models via Self-guided Adaptive Relevance Labeling","date":"2024-02-21","arxiv_id":"2402.13542","repositories_listed":0,"syntology":null},{"url":null,"slug":"have-seen-me-before-automating-dataset","title":"Automating Dataset Updates Towards Reliable and Timely Evaluation of Large Language Models","date":"2024-02-19","arxiv_id":"2402.11894","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-uncertainty-aware-language-agent","title":"Towards Uncertainty-Aware Language Agent","date":"2024-01-25","arxiv_id":"2401.14016","repositories_listed":0,"syntology":null},{"url":null,"slug":"llama-beyond-english-an-empirical-study-on","title":"LLaMA Beyond English: An Empirical Study on Language Capability Transfer","date":"2024-01-02","arxiv_id":"2401.01055","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-impact-of-prompting-persona-and","title":"Assessing the Impact of Prompting Methods on ChatGPT's Mathematical Capabilities","date":"2023-12-22","arxiv_id":"2312.15006","repositories_listed":0,"syntology":null},{"url":null,"slug":"yayi-2-multilingual-open-source-large","title":"YAYI 2: Multilingual Open-Source Large Language Models","date":"2023-12-22","arxiv_id":"2312.14862","repositories_listed":0,"syntology":null},{"url":null,"slug":"academicgpt-empowering-academic-research","title":"AcademicGPT: Empowering Academic Research","date":"2023-11-21","arxiv_id":"2311.12315","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-data-contamination-in-modern","title":"Investigating Data Contamination in Modern Benchmarks for Large Language Models","date":"2023-11-16","arxiv_id":"2311.09783","repositories_listed":0,"syntology":null},{"url":null,"slug":"psybench-a-balanced-and-in-depth","title":"ConceptPsy:A Benchmark Suite with Conceptual Comprehensiveness in Psychology","date":"2023-11-16","arxiv_id":"2311.09861","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-alignment-ceiling-objective-mismatch-in","title":"The Alignment Ceiling: Objective Mismatch in Reinforcement Learning from Human Feedback","date":"2023-10-31","arxiv_id":"2311.00168","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacherlm-teaching-to-fish-rather-than-giving","title":"TeacherLM: Teaching to Fish Rather Than Giving the Fish, Language Modeling Likewise","date":"2023-10-29","arxiv_id":"2310.19019","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-large-language-models-using-an","title":"Evaluation of large language models using an Indian language LGBTI+ lexicon","date":"2023-10-26","arxiv_id":"2310.17787","repositories_listed":0,"syntology":null},{"url":null,"slug":"irreducible-curriculum-for-language-model","title":"Irreducible Curriculum for Language Model Pretraining","date":"2023-10-23","arxiv_id":"2310.15389","repositories_listed":0,"syntology":null},{"url":null,"slug":"take-a-step-back-evoking-reasoning-via","title":"Take a Step Back: Evoking Reasoning via Abstraction in Large Language Models","date":"2023-10-09","arxiv_id":"2310.06117","repositories_listed":0,"syntology":null},{"url":null,"slug":"pruning-large-language-models-via-accuracy","title":"Pruning Large Language Models via Accuracy Predictor","date":"2023-09-18","arxiv_id":"2309.09507","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-poison-of-alignment","title":"The Poison of Alignment","date":"2023-08-25","arxiv_id":"2308.13449","repositories_listed":0,"syntology":null},{"url":null,"slug":"let-s-do-a-thought-experiment-using","title":"Let's Do a Thought Experiment: Using Counterfactuals to Improve Moral Reasoning","date":"2023-06-25","arxiv_id":"2306.14308","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-progress-on-scalable-oversight-for","title":"Measuring Progress on Scalable Oversight for Large Language Models","date":"2022-11-04","arxiv_id":"2211.03540","repositories_listed":0,"syntology":null},{"url":"/paper/transcending-scaling-laws-with-0-1-extra","slug":"transcending-scaling-laws-with-0-1-extra","title":"Transcending Scaling Laws with 0.1% Extra Compute","date":"2022-10-20","arxiv_id":"2210.11399","repositories_listed":0,"syntology":null}],"record_sha256":"719fe351b71dcf3ff1f599e8ed9e8d42b2f407ae0d199875a1d94bf166960b83","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}