{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/format-subject","entry":"format_subject","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":26,"n_papers_ran":25,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":4,"n_samples_ran":3,"n_samples_fingerprinted":3,"n_places":26,"n_places_pointer_only":11,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":2,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.15491","paper":"/paper/arxiv-2605-15491","title":"Ghosted Layers: Unconstrained Activation Alignment for Recovering Layer-Pruned LLMs","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"chenxinrui-tsinghua/LinearPatch","path":"eval/mmlu_eval.py","file_url":"https://github.com/chenxinrui-tsinghua/LinearPatch/blob/HEAD/eval/mmlu_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2602.10718","paper":"/paper/arxiv-2602-10718","title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"meituan-longcat/SGLang-FluentLLM","path":"benchmark/mmlu/bench_sglang.py","file_url":"https://github.com/meituan-longcat/SGLang-FluentLLM/blob/HEAD/benchmark/mmlu/bench_sglang.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2601.16503","paper":"/paper/arxiv-2601-16503","title":"MRAG: Benchmarking Retrieval-Augmented Generation for Bio-medicine","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"hendrycks/test","path":"evaluate.py","file_url":"https://github.com/hendrycks/test/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2505.24449","paper":"/paper/when-large-multimodal-models-confront","title":"When Large Multimodal Models Confront Evolving Knowledge:Challenges and Pathways","date":"2025-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pjlab-sys4nlp/llama-moe","path":"smoe/entrypoint/eval/eval_mmlu_moe_0.py","file_url":"https://github.com/pjlab-sys4nlp/llama-moe/blob/HEAD/smoe/entrypoint/eval/eval_mmlu_moe_0.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2410.21352","paper":"/paper/llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AboveParadise/LLMCBench","path":"evaluate_mmlu.py","file_url":"https://github.com/AboveParadise/LLMCBench/blob/HEAD/evaluate_mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2410.20745","paper":"/paper/shopping-mmlu-a-massive-multi-task-online","title":"Shopping MMLU: A Massive Multi-Task Online Shopping Benchmark for Large Language Models","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KL4805/ShoppingMMLU","path":"task_wise_eval/utils.py","file_url":"https://github.com/KL4805/ShoppingMMLU/blob/HEAD/task_wise_eval/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"83fbf53277c007f3","mcp_get_code":{"code_sha256":"83fbf53277c007f3"}},{"arxiv_id":"2410.04199","paper":"/paper/longgenbench-long-context-generation","title":"LongGenBench: Long-context Generation Benchmark","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2408.10284","paper":"/paper/adapmoe-adaptive-sensitivity-based-expert","title":"AdapMoE: Adaptive Sensitivity-based Expert Gating and Management for Efficient MoE Inference","date":"2024-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2407.16286","paper":"/paper/a-deeper-look-at-depth-pruning-of-llms","title":"A deeper look at depth pruning of LLMs","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shoaibahmed/llm_depth_pruning","path":"evals/mmlu.py","file_url":"https://github.com/shoaibahmed/llm_depth_pruning/blob/HEAD/evals/mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2406.18665","paper":"/paper/routellm-learning-to-route-llms-with","title":"RouteLLM: Learning to Route LLMs with Preference Data","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lm-sys/routellm","path":"routellm/evals/mmlu/generate_responses.py","file_url":"https://github.com/lm-sys/routellm/blob/HEAD/routellm/evals/mmlu/generate_responses.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2406.13236","paper":"/paper/data-contamination-can-cross-language","title":"Data Contamination Can Cross Language Barriers","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShangDataLab/Deep-Contam","path":"inject/run_sft_mmlu.py","file_url":"https://github.com/ShangDataLab/Deep-Contam/blob/HEAD/inject/run_sft_mmlu.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"793a8b6496415457","mcp_get_code":{"code_sha256":"793a8b6496415457"}},{"arxiv_id":"2406.01721","paper":"/paper/rotation-and-permutation-for-advanced-outlier","title":"DuQuant: Distributing Outliers via Dual Transformation Makes Stronger Quantized LLMs","date":"2024-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Hsu1023/DuQuant","path":"mmlu_eval.py","file_url":"https://github.com/Hsu1023/DuQuant/blob/HEAD/mmlu_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2404.02078","paper":"/paper/advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openbmb/eurus","path":"eval/mmlu/evaluate_mmlu.py","file_url":"https://github.com/openbmb/eurus/blob/HEAD/eval/mmlu/evaluate_mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2403.01244","paper":"/paper/mitigating-catastrophic-forgetting-in-large","title":"Mitigating Catastrophic Forgetting in Large Language Models with Self-Synthesized Rehearsal","date":"2024-03-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DeepLearnXMU/SSR","path":"mmlu_test/evaluate.py","file_url":"https://github.com/DeepLearnXMU/SSR/blob/HEAD/mmlu_test/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2402.18243","paper":"/paper/learning-or-self-aligning-rethinking","title":"Learning or Self-aligning? Rethinking Instruction Fine-tuning","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renmengjie7/self-aligning","path":"eval/my_benchmark_eval.py","file_url":"https://github.com/renmengjie7/self-aligning/blob/HEAD/eval/my_benchmark_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2401.06469","paper":"/paper/batch-icl-effective-efficient-and-order","title":"Batch-ICL: Effective, Efficient, and Order-Agnostic In-Context Learning","date":"2024-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cardinalere/batch-icl","path":"Batch-ICL-multi_epoch.py","file_url":"https://github.com/cardinalere/batch-icl/blob/HEAD/Batch-ICL-multi_epoch.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2312.11875","paper":"/paper/sparse-is-enough-in-fine-tuning-pre-trained","title":"Sparse is Enough in Fine-tuning Pre-trained Large Language Models","date":"2023-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"song-wx/SIFT","path":"exp/mmlu/eval_mmlu.py","file_url":"https://github.com/song-wx/SIFT/blob/HEAD/exp/mmlu/eval_mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2312.07104","paper":"/paper/efficiently-programming-large-language-models","title":"SGLang: Efficient Execution of Structured Language Model Programs","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tginart/sglang","path":"benchmark/mmlu/bench_sglang.py","file_url":"https://github.com/tginart/sglang/blob/HEAD/benchmark/mmlu/bench_sglang.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2312.00700","paper":"/paper/gift-generative-interpretable-fine-tuning","title":"Generative Parameter-Efficient Fine-Tuning","date":"2023-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"savadikarc/gift","path":"language_modeling/math_code_instruct/eval_mmlu.py","file_url":"https://github.com/savadikarc/gift/blob/HEAD/language_modeling/math_code_instruct/eval_mmlu.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"793a8b6496415457","mcp_get_code":{"code_sha256":"793a8b6496415457"}},{"arxiv_id":"2311.09677","paper":"/paper/r-tuning-teaching-large-language-models-to","title":"R-Tuning: Instructing Large Language Models to Say `I Don't Know'","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/r-tuning","path":"evaluation/MMLU/evaluate.py","file_url":"https://github.com/shizhediao/r-tuning/blob/HEAD/evaluation/MMLU/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2310.05736","paper":"/paper/llmlingua-compressing-prompts-for-accelerated","title":"LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FranxYao/chain-of-thought-hub","path":"MMLU/run_mmlu_llama.py","file_url":"https://github.com/FranxYao/chain-of-thought-hub/blob/HEAD/MMLU/run_mmlu_llama.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2310.01651","paper":"/paper/fool-your-vision-and-language-model-with","title":"Fool Your (Vision and) Language Model With Embarrassingly Simple Permutations","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2310.00297","paper":"/paper/understanding-in-context-learning-from","title":"Understanding In-Context Learning from Repetitions","date":"2023-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"elliottyan/understand-icl-from-repetition","path":"analyze_icl_rep/analyze_mmlu_label_space.py","file_url":"https://github.com/elliottyan/understand-icl-from-repetition/blob/HEAD/analyze_icl_rep/analyze_mmlu_label_space.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"da9a2d07dd8aa31a","mcp_get_code":{"code_sha256":"da9a2d07dd8aa31a"}},{"arxiv_id":"2305.17306","paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"franxyao/chain-of-thought-hub","path":"MMLU/run_mmlu_llama.py","file_url":"https://github.com/franxyao/chain-of-thought-hub/blob/HEAD/MMLU/run_mmlu_llama.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2304.12986","paper":"/paper/measuring-massive-multitask-chinese","title":"Measuring Massive Multitask Chinese Understanding","date":"2023-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}},{"arxiv_id":"2009.03300","paper":"/paper/measuring-massive-multitask-language","title":"Measuring Massive Multitask Language Understanding","date":"2020-09-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ollmer/mmlu","path":"evaluate.py","file_url":"https://github.com/ollmer/mmlu/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6ab745408cb8648b","mcp_get_code":{"code_sha256":"6ab745408cb8648b"}}]}