{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/parse-multi-choice-response","entry":"parse_multi_choice_response","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":13,"n_papers_ran":2,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":5,"n_samples_ran":2,"n_samples_fingerprinted":1,"n_places":13,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":0,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2509.19003","paper":"/paper/arxiv-2509-19003","title":"Unveiling Chain of Step Reasoning for Vision-Language Models with Fine-grained Rewards","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"baaivision/CoS","path":"eval/mmmu/eval_utils.py","file_url":"https://github.com/baaivision/CoS/blob/HEAD/eval/mmmu/eval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"2505.19028","paper":"/paper/infochartqa-a-benchmark-for-multimodal","title":"InfoChartQA: A Benchmark for Multimodal Question Answering on Infographic Charts","date":"2025-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cooldawnant/infochartqa","path":"eval/compare_multiple.py","file_url":"https://github.com/cooldawnant/infochartqa/blob/HEAD/eval/compare_multiple.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d86d4c52f0d2788b","mcp_get_code":{"code_sha256":"d86d4c52f0d2788b"}},{"arxiv_id":"2502.00698","paper":"/paper/mm-iq-benchmarking-human-like-abstraction-and-1","title":"MM-IQ: Benchmarking Human-Like Abstraction and Reasoning in Multimodal Models","date":"2025-02-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AceCHQ/MMIQ","path":"mmiq/utils/eval_utils.py","file_url":"https://github.com/AceCHQ/MMIQ/blob/HEAD/mmiq/utils/eval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c2cf50db9a54f717","mcp_get_code":{"code_sha256":"c2cf50db9a54f717"}},{"arxiv_id":"2412.12359","paper":"/paper/visual-instruction-tuning-with-500x-fewer","title":"LLaVA Steering: Visual Instruction Tuning with 500x Fewer Parameters through Modality Linear Representation-Steering","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bibisbar/LLaVA-Steering","path":"tinyllava/eval/model_vqa_mmmu.py","file_url":"https://github.com/bibisbar/LLaVA-Steering/blob/HEAD/tinyllava/eval/model_vqa_mmmu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"2410.16236","paper":"/paper/llava-kd-a-framework-of-distilling-multimodal","title":"LLaVA-KD: A Framework of Distilling Multimodal Large Language Models","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Fantasyele/LLaVA-KD","path":"llavakd/eval/model_vqa_mmmu.py","file_url":"https://github.com/Fantasyele/LLaVA-KD/blob/HEAD/llavakd/eval/model_vqa_mmmu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"2410.05269","paper":"/paper/data-advisor-dynamic-data-curation-for-safety","title":"Data Advisor: Dynamic Data Curation for Safety Alignment of Large Language Models","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"feiwang96/Data-Advisor","path":"eval/parse_prediction.py","file_url":"https://github.com/feiwang96/Data-Advisor/blob/HEAD/eval/parse_prediction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"2406.15279","paper":"/paper/cross-modality-safety-alignment","title":"Cross-Modality Safety Alignment","date":"2024-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sinwang20/siuo","path":"eval/mcqa-eval.py","file_url":"https://github.com/sinwang20/siuo/blob/HEAD/eval/mcqa-eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8f778695ae1465d2","mcp_get_code":{"code_sha256":"8f778695ae1465d2"}},{"arxiv_id":"2406.14852","paper":"/paper/is-a-picture-worth-a-thousand-words-delving","title":"Is A Picture Worth A Thousand Words? Delving Into Spatial Reasoning for Vision Language Models","date":"2024-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BAAI-DCAI/Bunny","path":"bunny/eval/model_vqa_mmmu.py","file_url":"https://github.com/BAAI-DCAI/Bunny/blob/HEAD/bunny/eval/model_vqa_mmmu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"2406.09411","paper":"/paper/muirbench-a-comprehensive-benchmark-for","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"muirbench/MuirBench","path":"eval/utils/postprocess.py","file_url":"https://github.com/muirbench/MuirBench/blob/HEAD/eval/utils/postprocess.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b745b9c89ddb6cf5","mcp_get_code":{"code_sha256":"b745b9c89ddb6cf5"}},{"arxiv_id":"2402.14289","paper":"/paper/tinyllava-a-framework-of-small-scale-large","title":"TinyLLaVA: A Framework of Small-scale Large Multimodal Models","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dlcv-buaa/tinyllavabench","path":"tinyllava/eval/model_vqa_mmmu.py","file_url":"https://github.com/dlcv-buaa/tinyllavabench/blob/HEAD/tinyllava/eval/model_vqa_mmmu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"2402.11530","paper":"/paper/efficient-multimodal-learning-from-data","title":"Efficient Multimodal Learning from Data-centric Perspective","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"baai-dcai/bunny","path":"bunny/eval/model_vqa_mmmu.py","file_url":"https://github.com/baai-dcai/bunny/blob/HEAD/bunny/eval/model_vqa_mmmu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"2311.10774","paper":"/paper/mmc-advancing-multimodal-chart-understanding","title":"MMC: Advancing Multimodal Chart Understanding with Large-scale Instruction Tuning","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fuxiaoliu/mmc","path":"Eval/utils.py","file_url":"https://github.com/fuxiaoliu/mmc/blob/HEAD/Eval/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}},{"arxiv_id":"Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","paper":null,"title":"arXiv:Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenGVLab/PVC","path":"eval/mmmu/eval_utils.py","file_url":"https://github.com/OpenGVLab/PVC/blob/HEAD/eval/mmmu/eval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3018307f9301da1","mcp_get_code":{"code_sha256":"f3018307f9301da1"}}]}