{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/format-question","entry":"format_question","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":11,"n_papers_ran":6,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":14,"n_samples_ran":9,"n_samples_fingerprinted":1,"n_places":14,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":7,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.05973","paper":"/paper/arxiv-2605-05973","title":"Towards Reliable LLM Evaluation: Correcting the Winner's Curse in Adaptive Benchmarking","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"jznmsl/siren","path":"03_analysis_scripts/study_e_multimodel_api.py","file_url":"https://github.com/jznmsl/siren/blob/HEAD/03_analysis_scripts/study_e_multimodel_api.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7dc3245651ee6f5e","mcp_get_code":{"code_sha256":"7dc3245651ee6f5e"}},{"arxiv_id":"2604.04356","paper":"/paper/arxiv-2604-04356","title":"REAM: Merging Improves Pruning of Experts in LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zai-org/glm-simple-evals","path":"evals/gpqa_eval.py","file_url":"https://github.com/zai-org/glm-simple-evals/blob/HEAD/evals/gpqa_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"aaef2696f445a0a2","mcp_get_code":{"code_sha256":"aaef2696f445a0a2"}},{"arxiv_id":"2503.08688","paper":"/paper/randomness-not-representation-the","title":"Randomness, Not Representation: The Unreliability of Evaluating Cultural Alignment in LLMs","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":"archive","repo":"ariba-k/llm-cultural-alignment-evaluation","path":"extrapolability/run_extrapolability.py","file_url":"https://github.com/ariba-k/llm-cultural-alignment-evaluation/blob/HEAD/extrapolability/run_extrapolability.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c9ff479d96e5f207","mcp_get_code":{"code_sha256":"c9ff479d96e5f207"}},{"arxiv_id":"2503.07459","paper":"/paper/medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gersteinlab/medagents-benchmark","path":"plots/annotate_reasoning_depth.py","file_url":"https://github.com/gersteinlab/medagents-benchmark/blob/HEAD/plots/annotate_reasoning_depth.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4eb21a43f63b6a84","mcp_get_code":{"code_sha256":"4eb21a43f63b6a84"}},{"arxiv_id":"2406.07302","paper":"/paper/bertaqa-how-much-do-language-models-know","title":"BertaQA: How Much Do Language Models Know About Local Culture?","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"juletx/BertaQA","path":"openai/evaluate.py","file_url":"https://github.com/juletx/BertaQA/blob/HEAD/openai/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7151aae860cbcdd0","mcp_get_code":{"code_sha256":"7151aae860cbcdd0"}},{"arxiv_id":"2311.09677","paper":"/paper/r-tuning-teaching-large-language-models-to","title":"R-Tuning: Instructing Large Language Models to Say `I Don't Know'","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/r-tuning","path":"evaluation/FEVER/evaluate.py","file_url":"https://github.com/shizhediao/r-tuning/blob/HEAD/evaluation/FEVER/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c62af2431eed8398","mcp_get_code":{"code_sha256":"c62af2431eed8398"}},{"arxiv_id":"2311.09677","paper":"/paper/r-tuning-teaching-large-language-models-to","title":"R-Tuning: Instructing Large Language Models to Say `I Don't Know'","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/r-tuning","path":"evaluation/HaluEvalQA/evaluate.py","file_url":"https://github.com/shizhediao/r-tuning/blob/HEAD/evaluation/HaluEvalQA/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ed702608cd107f17","mcp_get_code":{"code_sha256":"ed702608cd107f17"}},{"arxiv_id":"2311.09677","paper":"/paper/r-tuning-teaching-large-language-models-to","title":"R-Tuning: Instructing Large Language Models to Say `I Don't Know'","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/r-tuning","path":"evaluation/HotpotQA/evaluate.py","file_url":"https://github.com/shizhediao/r-tuning/blob/HEAD/evaluation/HotpotQA/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8ffd4f62ece5b520","mcp_get_code":{"code_sha256":"8ffd4f62ece5b520"}},{"arxiv_id":"2311.09677","paper":"/paper/r-tuning-teaching-large-language-models-to","title":"R-Tuning: Instructing Large Language Models to Say `I Don't Know'","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhediao/r-tuning","path":"evaluation/WiCE/evaluate.py","file_url":"https://github.com/shizhediao/r-tuning/blob/HEAD/evaluation/WiCE/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c3a776c309a37593","mcp_get_code":{"code_sha256":"c3a776c309a37593"}},{"arxiv_id":"2310.13132","paper":"/paper/ask-me-in-english-instead-cross-lingual","title":"Better to Ask in English: Cross-Lingual Evaluation of Large Language Models for Healthcare Queries","date":"2023-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"claws-lab/XLingEval","path":"consistency/consistency_get_medalpaca_answer.py","file_url":"https://github.com/claws-lab/XLingEval/blob/HEAD/consistency/consistency_get_medalpaca_answer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b4bf5a8362f1b8a1","mcp_get_code":{"code_sha256":"b4bf5a8362f1b8a1"}},{"arxiv_id":"2308.07317","paper":"/paper/platypus-quick-cheap-and-powerful-refinement","title":"Platypus: Quick, Cheap, and Powerful Refinement of LLMs","date":"2023-08-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"arielnlee/Platypus","path":"data_pipeline/reclor_formatted.py","file_url":"https://github.com/arielnlee/Platypus/blob/HEAD/data_pipeline/reclor_formatted.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb28b7e4bde799c6","mcp_get_code":{"code_sha256":"cb28b7e4bde799c6"}},{"arxiv_id":"2305.18395","paper":"/paper/knowledge-augmented-reasoning-distillation-1","title":"Knowledge-Augmented Reasoning Distillation for Small Language Models in Knowledge-Intensive Tasks","date":"2023-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nardien/kard","path":"preprocess_cot.py","file_url":"https://github.com/nardien/kard/blob/HEAD/preprocess_cot.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ab34cb918863ef3c","mcp_get_code":{"code_sha256":"ab34cb918863ef3c"}},{"arxiv_id":"2025.findings-acl.321","paper":null,"title":"arXiv:2025.findings-acl.321","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"gzy02/FRAG","path":"FRAG/src/FRAG.py","file_url":"https://github.com/gzy02/FRAG/blob/HEAD/FRAG/src/FRAG.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a417e5858425502e","mcp_get_code":{"code_sha256":"a417e5858425502e"}},{"arxiv_id":"2023.findings-emnlp.1023","paper":null,"title":"arXiv:2023.findings-emnlp.1023","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HKUST-KnowComp/QaDynamics","path":"src/Data_generation/generate_from_CWWV.py","file_url":"https://github.com/HKUST-KnowComp/QaDynamics/blob/HEAD/src/Data_generation/generate_from_CWWV.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6a37a5935f56b90a","mcp_get_code":{"code_sha256":"6a37a5935f56b90a"}}]}