{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/format-example","entry":"format_example","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":43,"n_papers_ran":33,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":34,"n_samples_ran":20,"n_samples_fingerprinted":2,"n_places":50,"n_places_pointer_only":19,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":9,"ran_fixture":0,"ran":11,"unverified":14},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.15491","paper":"/paper/arxiv-2605-15491","title":"Ghosted Layers: Unconstrained Activation Alignment for Recovering Layer-Pruned LLMs","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"chenxinrui-tsinghua/LinearPatch","path":"eval/mmlu_eval.py","file_url":"https://github.com/chenxinrui-tsinghua/LinearPatch/blob/HEAD/eval/mmlu_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2604.04356","paper":"/paper/arxiv-2604-04356","title":"REAM: Merging Improves Pruning of Experts in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zai-org/glm-simple-evals","path":"evals/mmlu_pro_eval.py","file_url":"https://github.com/zai-org/glm-simple-evals/blob/HEAD/evals/mmlu_pro_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"24ff49e5a1d92207","mcp_get_code":{"code_sha256":"24ff49e5a1d92207"}},{"arxiv_id":"2602.10718","paper":"/paper/arxiv-2602-10718","title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"meituan-longcat/SGLang-FluentLLM","path":"benchmark/mmlu/bench_sglang.py","file_url":"https://github.com/meituan-longcat/SGLang-FluentLLM/blob/HEAD/benchmark/mmlu/bench_sglang.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2601.16503","paper":"/paper/arxiv-2601-16503","title":"MRAG: Benchmarking Retrieval-Augmented Generation for Bio-medicine","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"hendrycks/test","path":"evaluate.py","file_url":"https://github.com/hendrycks/test/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2510.15346","paper":"/paper/arxiv-2510-15346","title":"When to Ensemble: Identifying Token-Level Points for Stable and Fast LLM Ensembling WHEN TO ENSEMBLE: IDENTIFYING TOKEN-LEVEL POINTS FOR STABLE AND FAST LLM ENSEMBLING","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"yoon6503/SAFE","path":"run_safe.py","file_url":"https://github.com/yoon6503/SAFE/blob/HEAD/run_safe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c8947f39a6f3b186","mcp_get_code":{"code_sha256":"c8947f39a6f3b186"}},{"arxiv_id":"2510.15346","paper":"/paper/arxiv-2510-15346","title":"When to Ensemble: Identifying Token-Level Points for Stable and Fast LLM Ensembling WHEN TO ENSEMBLE: IDENTIFYING TOKEN-LEVEL POINTS FOR STABLE AND FAST LLM ENSEMBLING","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"yoon6503/SAFE","path":"run_safe_mmlu.py","file_url":"https://github.com/yoon6503/SAFE/blob/HEAD/run_safe_mmlu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"914abaca965ea62f","mcp_get_code":{"code_sha256":"914abaca965ea62f"}},{"arxiv_id":"2507.11953","paper":null,"title":"arXiv:2507.11953","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"QwenLM/Qwen","path":"eval/evaluate_ceval.py","file_url":"https://github.com/QwenLM/Qwen/blob/HEAD/eval/evaluate_ceval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"15a59658a1a9f53d","mcp_get_code":{"code_sha256":"15a59658a1a9f53d"}},{"arxiv_id":"2505.24449","paper":"/paper/when-large-multimodal-models-confront","title":"When Large Multimodal Models Confront Evolving Knowledge:Challenges and Pathways","date":"2025-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pjlab-sys4nlp/llama-moe","path":"smoe/entrypoint/eval/eval_mmlu_moe_0.py","file_url":"https://github.com/pjlab-sys4nlp/llama-moe/blob/HEAD/smoe/entrypoint/eval/eval_mmlu_moe_0.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2505.16690","paper":"/paper/your-pre-trained-llm-is-secretly-an","title":"Your Pre-trained LLM is Secretly an Unsupervised Confidence Calibrator","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ml-stat-Sustech/Disagreement-Aware-Calibration","path":"common/datasets.py","file_url":"https://github.com/ml-stat-Sustech/Disagreement-Aware-Calibration/blob/HEAD/common/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"718d1ba9049ace00","mcp_get_code":{"code_sha256":"718d1ba9049ace00"}},{"arxiv_id":"2411.13504","paper":"/paper/disentangling-memory-and-reasoning-ability-in","title":"Disentangling Memory and Reasoning Ability in Large Language Models","date":"2024-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mingyuj666/disentangling-memory-and-reasoning","path":"load_data/data_agent.py","file_url":"https://github.com/mingyuj666/disentangling-memory-and-reasoning/blob/HEAD/load_data/data_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b82b37c35d49aafd","mcp_get_code":{"code_sha256":"b82b37c35d49aafd"}},{"arxiv_id":"2410.21352","paper":"/paper/llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AboveParadise/LLMCBench","path":"evaluate_mmlu.py","file_url":"https://github.com/AboveParadise/LLMCBench/blob/HEAD/evaluate_mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2410.21352","paper":"/paper/llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AboveParadise/LLMCBench","path":"evaluate_advglue.py","file_url":"https://github.com/AboveParadise/LLMCBench/blob/HEAD/evaluate_advglue.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"984d3567b5126f65","mcp_get_code":{"code_sha256":"984d3567b5126f65"}},{"arxiv_id":"2410.21352","paper":"/paper/llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AboveParadise/LLMCBench","path":"evaluate_mnli.py","file_url":"https://github.com/AboveParadise/LLMCBench/blob/HEAD/evaluate_mnli.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7538653cf2565390","mcp_get_code":{"code_sha256":"7538653cf2565390"}},{"arxiv_id":"2410.21352","paper":"/paper/llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AboveParadise/LLMCBench","path":"evaluate_qnli.py","file_url":"https://github.com/AboveParadise/LLMCBench/blob/HEAD/evaluate_qnli.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dafa37632f3f91bf","mcp_get_code":{"code_sha256":"dafa37632f3f91bf"}},{"arxiv_id":"2410.20745","paper":"/paper/shopping-mmlu-a-massive-multi-task-online","title":"Shopping MMLU: A Massive Multi-Task Online Shopping Benchmark for Large Language Models","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KL4805/ShoppingMMLU","path":"task_wise_eval/utils.py","file_url":"https://github.com/KL4805/ShoppingMMLU/blob/HEAD/task_wise_eval/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c78d363cf62ef64c","mcp_get_code":{"code_sha256":"c78d363cf62ef64c"}},{"arxiv_id":"2410.04223","paper":"/paper/multimodal-large-language-models-for-inverse","title":"Multimodal Large Language Models for Inverse Molecular Design with Retrosynthetic Planning","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liugangcode/Llamole","path":"launch.py","file_url":"https://github.com/liugangcode/Llamole/blob/HEAD/launch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c781f40e814fc9a","mcp_get_code":{"code_sha256":"3c781f40e814fc9a"}},{"arxiv_id":"2409.02257","paper":"/paper/mmlu-pro-evaluating-higher-order-reasoning","title":"MMLU-Pro+: Evaluating Higher-Order Reasoning and Shortcut Learning in LLMs","date":"2024-09-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"asgsaeid/mmlu-pro-plus","path":"evaluate_from_api.py","file_url":"https://github.com/asgsaeid/mmlu-pro-plus/blob/HEAD/evaluate_from_api.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"df80a0e9f84176ff","mcp_get_code":{"code_sha256":"df80a0e9f84176ff"}},{"arxiv_id":"2408.10284","paper":"/paper/adapmoe-adaptive-sensitivity-based-expert","title":"AdapMoE: Adaptive Sensitivity-based Expert Gating and Management for Efficient MoE Inference","date":"2024-08-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2408.07888","paper":"/paper/fine-tuning-large-language-models-with-human","title":"Evaluating Fine-Tuning Efficiency of Human-Inspired Learning Strategies in Medical Question Answering","date":"2024-08-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Oxford-AI-for-Society/human-learning-strategies","path":"training/fine_tuning/shared_utils.py","file_url":"https://github.com/Oxford-AI-for-Society/human-learning-strategies/blob/HEAD/training/fine_tuning/shared_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"6ea1d20b1706f31c","mcp_get_code":{"code_sha256":"6ea1d20b1706f31c"}},{"arxiv_id":"2407.16286","paper":"/paper/a-deeper-look-at-depth-pruning-of-llms","title":"A deeper look at depth pruning of LLMs","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shoaibahmed/llm_depth_pruning","path":"evals/mmlu.py","file_url":"https://github.com/shoaibahmed/llm_depth_pruning/blob/HEAD/evals/mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2406.18665","paper":"/paper/routellm-learning-to-route-llms-with","title":"RouteLLM: Learning to Route LLMs with Preference Data","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lm-sys/routellm","path":"routellm/evals/mmlu/generate_responses.py","file_url":"https://github.com/lm-sys/routellm/blob/HEAD/routellm/evals/mmlu/generate_responses.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2406.16330","paper":"/paper/pruning-via-merging-compressing-llms-via","title":"Pruning via Merging: Compressing LLMs via Manifold Alignment Based Layer Merging","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sempraety/pruning-via-merging","path":"pipeline.py","file_url":"https://github.com/sempraety/pruning-via-merging/blob/HEAD/pipeline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e65eff7d9e1ab5f3","mcp_get_code":{"code_sha256":"e65eff7d9e1ab5f3"}},{"arxiv_id":"2406.13948","paper":"/paper/citygpt-empowering-urban-spatial-cognition-of","title":"CityGPT: Empowering Urban Spatial Cognition of Large Language Models","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsinghua-fib-lab/citygpt","path":"evaluate/city_eval/run_eval.py","file_url":"https://github.com/tsinghua-fib-lab/citygpt/blob/HEAD/evaluate/city_eval/run_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6d6626dd2774d1c","mcp_get_code":{"code_sha256":"f6d6626dd2774d1c"}},{"arxiv_id":"2406.13945","paper":"/paper/citybench-evaluating-the-capabilities-of","title":"CityBench: Evaluating the Capabilities of Large Language Models for Urban Tasks","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsinghua-fib-lab/citybench","path":"citybench/geoqa/run_eval.py","file_url":"https://github.com/tsinghua-fib-lab/citybench/blob/HEAD/citybench/geoqa/run_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fbde55d407dfbc1f","mcp_get_code":{"code_sha256":"fbde55d407dfbc1f"}},{"arxiv_id":"2406.13236","paper":"/paper/data-contamination-can-cross-language","title":"Data Contamination Can Cross Language Barriers","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShangDataLab/Deep-Contam","path":"inject/run_sft_mmlu.py","file_url":"https://github.com/ShangDataLab/Deep-Contam/blob/HEAD/inject/run_sft_mmlu.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"16d87a980525188f","mcp_get_code":{"code_sha256":"16d87a980525188f"}},{"arxiv_id":"2406.06331","paper":"/paper/medexqa-medical-question-answering-benchmark","title":"MedExQA: Medical Question Answering Benchmark with Multiple Explanations","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"knowlab/medexqa","path":"evaluate_MedExQA.py","file_url":"https://github.com/knowlab/medexqa/blob/HEAD/evaluate_MedExQA.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1a8789ccdb8f2890","mcp_get_code":{"code_sha256":"1a8789ccdb8f2890"}},{"arxiv_id":"2406.06331","paper":"/paper/medexqa-medical-question-answering-benchmark","title":"MedExQA: Medical Question Answering Benchmark with Multiple Explanations","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"knowlab/medexqa","path":"evaluate_pipe_MedExQA.py","file_url":"https://github.com/knowlab/medexqa/blob/HEAD/evaluate_pipe_MedExQA.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2836f4edcd88af80","mcp_get_code":{"code_sha256":"2836f4edcd88af80"}},{"arxiv_id":"2406.01721","paper":"/paper/rotation-and-permutation-for-advanced-outlier","title":"DuQuant: Distributing Outliers via Dual Transformation Makes Stronger Quantized LLMs","date":"2024-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Hsu1023/DuQuant","path":"mmlu_eval.py","file_url":"https://github.com/Hsu1023/DuQuant/blob/HEAD/mmlu_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2406.01574","paper":"/paper/mmlu-pro-a-more-robust-and-challenging-multi","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","date":"2024-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tiger-ai-lab/mmlu-pro","path":"evaluate_from_api.py","file_url":"https://github.com/tiger-ai-lab/mmlu-pro/blob/HEAD/evaluate_from_api.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1b3ff879d7f6d22b","mcp_get_code":{"code_sha256":"1b3ff879d7f6d22b"}},{"arxiv_id":"2405.11966","paper":"/paper/multiple-choice-questions-are-efficient-and","title":"Multiple-Choice Questions are Efficient and Robust LLM Evaluators","date":"2024-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"geralt-targaryen/mc-evaluation","path":"dataset_mc.py","file_url":"https://github.com/geralt-targaryen/mc-evaluation/blob/HEAD/dataset_mc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3d5d3f0243075f18","mcp_get_code":{"code_sha256":"3d5d3f0243075f18"}},{"arxiv_id":"2404.09492","paper":"/paper/bridging-the-gap-between-different","title":"Bridging the Gap between Different Vocabularies for LLM Ensemble","date":"2024-04-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xydaytoy/eva","path":"ensemble/eva_multi.py","file_url":"https://github.com/xydaytoy/eva/blob/HEAD/ensemble/eva_multi.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef417469242d0e7f","mcp_get_code":{"code_sha256":"ef417469242d0e7f"}},{"arxiv_id":"2404.02078","paper":"/paper/advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openbmb/eurus","path":"eval/mmlu/evaluate_mmlu.py","file_url":"https://github.com/openbmb/eurus/blob/HEAD/eval/mmlu/evaluate_mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2403.01244","paper":"/paper/mitigating-catastrophic-forgetting-in-large","title":"Mitigating Catastrophic Forgetting in Large Language Models with Self-Synthesized Rehearsal","date":"2024-03-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DeepLearnXMU/SSR","path":"mmlu_test/evaluate.py","file_url":"https://github.com/DeepLearnXMU/SSR/blob/HEAD/mmlu_test/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2402.18243","paper":"/paper/learning-or-self-aligning-rethinking","title":"Learning or Self-aligning? Rethinking Instruction Fine-tuning","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renmengjie7/self-aligning","path":"eval/my_benchmark_eval.py","file_url":"https://github.com/renmengjie7/self-aligning/blob/HEAD/eval/my_benchmark_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"32b8aeb6f6c62b62","mcp_get_code":{"code_sha256":"32b8aeb6f6c62b62"}},{"arxiv_id":"2402.18243","paper":"/paper/learning-or-self-aligning-rethinking","title":"Learning or Self-aligning? Rethinking Instruction Fine-tuning","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renmengjie7/self-aligning","path":"eval/my_domain_eval.py","file_url":"https://github.com/renmengjie7/self-aligning/blob/HEAD/eval/my_domain_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"64ff6430c270b1c9","mcp_get_code":{"code_sha256":"64ff6430c270b1c9"}},{"arxiv_id":"2312.11875","paper":"/paper/sparse-is-enough-in-fine-tuning-pre-trained","title":"Sparse is Enough in Fine-tuning Pre-trained Large Language Models","date":"2023-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"song-wx/SIFT","path":"exp/mmlu/eval_mmlu.py","file_url":"https://github.com/song-wx/SIFT/blob/HEAD/exp/mmlu/eval_mmlu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2312.07104","paper":"/paper/efficiently-programming-large-language-models","title":"SGLang: Efficient Execution of Structured Language Model Programs","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tginart/sglang","path":"benchmark/mmlu/bench_sglang.py","file_url":"https://github.com/tginart/sglang/blob/HEAD/benchmark/mmlu/bench_sglang.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2312.00700","paper":"/paper/gift-generative-interpretable-fine-tuning","title":"Generative Parameter-Efficient Fine-Tuning","date":"2023-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"savadikarc/gift","path":"language_modeling/math_code_instruct/eval_mmlu.py","file_url":"https://github.com/savadikarc/gift/blob/HEAD/language_modeling/math_code_instruct/eval_mmlu.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dec15cc75848a98e","mcp_get_code":{"code_sha256":"dec15cc75848a98e"}},{"arxiv_id":"2311.09677","paper":"/paper/r-tuning-teaching-large-language-models-to","title":"R-Tuning: Instructing Large Language Models to Say `I Don't Know'","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/r-tuning","path":"evaluation/MMLU/evaluate.py","file_url":"https://github.com/shizhediao/r-tuning/blob/HEAD/evaluation/MMLU/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"434b759fa0cba061","mcp_get_code":{"code_sha256":"434b759fa0cba061"}},{"arxiv_id":"2310.09550","paper":"/paper/can-large-language-model-comprehend-ancient","title":"Can Large Language Model Comprehend Ancient Chinese? A Preliminary Test on ACLUE","date":"2023-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"isen-zhang/aclue","path":"src/utils.py","file_url":"https://github.com/isen-zhang/aclue/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7ad8ee11528e78b","mcp_get_code":{"code_sha256":"c7ad8ee11528e78b"}},{"arxiv_id":"2310.06245","paper":"/paper/we-are-what-we-repeatedly-do-inducing-and","title":"We are what we repeatedly do: Inducing and deploying habitual schemas in persona-based responses","date":"2023-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bkane2/habitual-response-generation","path":"src/generate_response.py","file_url":"https://github.com/bkane2/habitual-response-generation/blob/HEAD/src/generate_response.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"24893f346f97c656","mcp_get_code":{"code_sha256":"24893f346f97c656"}},{"arxiv_id":"2310.05736","paper":"/paper/llmlingua-compressing-prompts-for-accelerated","title":"LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FranxYao/chain-of-thought-hub","path":"MMLU/run_mmlu_llama.py","file_url":"https://github.com/FranxYao/chain-of-thought-hub/blob/HEAD/MMLU/run_mmlu_llama.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2306.06031","paper":"/paper/fingpt-open-source-financial-large-language","title":"FinGPT: Open-Source Financial Large Language Models","date":"2023-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ai4finance-foundation/finnlp","path":"finnlp/benchmarks/fiqa.py","file_url":"https://github.com/ai4finance-foundation/finnlp/blob/HEAD/finnlp/benchmarks/fiqa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1b801af1b3425cfa","mcp_get_code":{"code_sha256":"1b801af1b3425cfa"}},{"arxiv_id":"2305.17306","paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"franxyao/chain-of-thought-hub","path":"MMLU/run_mmlu_llama.py","file_url":"https://github.com/franxyao/chain-of-thought-hub/blob/HEAD/MMLU/run_mmlu_llama.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2304.12986","paper":"/paper/measuring-massive-multitask-chinese","title":"Measuring Massive Multitask Chinese Understanding","date":"2023-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Felixgithub2017/MMCU","path":"TestBloomz.py","file_url":"https://github.com/Felixgithub2017/MMCU/blob/HEAD/TestBloomz.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4f2a52c39dd4a34","mcp_get_code":{"code_sha256":"a4f2a52c39dd4a34"}},{"arxiv_id":"2210.02414","paper":"/paper/glm-130b-an-open-bilingual-pre-trained-model","title":"GLM-130B: An Open Bilingual Pre-trained Model","date":"2022-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jackaduma/ChatGLM-LoRA-RLHF-PyTorch","path":"cover_alpaca2jsonl.py","file_url":"https://github.com/jackaduma/ChatGLM-LoRA-RLHF-PyTorch/blob/HEAD/cover_alpaca2jsonl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1b801af1b3425cfa","mcp_get_code":{"code_sha256":"1b801af1b3425cfa"}},{"arxiv_id":"2009.03300","paper":"/paper/measuring-massive-multitask-language","title":"Measuring Massive Multitask Language Understanding","date":"2020-09-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ollmer/mmlu","path":"evaluate.py","file_url":"https://github.com/ollmer/mmlu/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd763eaf1ac287e7","mcp_get_code":{"code_sha256":"cd763eaf1ac287e7"}},{"arxiv_id":"2025.findings-emnlp.484","paper":null,"title":"arXiv:2025.findings-emnlp.484","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"UbiquitousLearning/DroidCall","path":"gen_complex_instructions.py","file_url":"https://github.com/UbiquitousLearning/DroidCall/blob/HEAD/gen_complex_instructions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2a66957120b8345c","mcp_get_code":{"code_sha256":"2a66957120b8345c"}},{"arxiv_id":"2025.findings-emnlp.484","paper":null,"title":"arXiv:2025.findings-emnlp.484","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"UbiquitousLearning/DroidCall","path":"gen_instructions.py","file_url":"https://github.com/UbiquitousLearning/DroidCall/blob/HEAD/gen_instructions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6544f5aba9b4a618","mcp_get_code":{"code_sha256":"6544f5aba9b4a618"}},{"arxiv_id":"2025.emnlp-main.67","paper":null,"title":"arXiv:2025.emnlp-main.67","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"isyuhaochen/RRC-DSCD","path":"code/sft/sft.py","file_url":"https://github.com/isyuhaochen/RRC-DSCD/blob/HEAD/code/sft/sft.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"27282e44111d98c4","mcp_get_code":{"code_sha256":"27282e44111d98c4"}}]}