{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/safe-equal","entry":"safe_equal","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":9,"n_papers_ran":2,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":3,"n_samples_ran":2,"n_samples_fingerprinted":1,"n_places":9,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":1,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2509.19003","paper":"/paper/arxiv-2509-19003","title":"Unveiling Chain of Step Reasoning for Vision-Language Models with Fine-grained Rewards","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"baaivision/CoS","path":"eval/mathvista/calculate_score.py","file_url":"https://github.com/baaivision/CoS/blob/HEAD/eval/mathvista/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e7a23ea3695f6437","mcp_get_code":{"code_sha256":"e7a23ea3695f6437"}},{"arxiv_id":"2410.17885","paper":"/paper/r-cot-reverse-chain-of-thought-problem","title":"R-CoT: Reverse Chain-of-Thought Problem Generation for Geometric Reasoning in Large Multimodal Models","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dle666/r-cot","path":"MathVista_eval/evaluation/calculate_score.py","file_url":"https://github.com/dle666/r-cot/blob/HEAD/MathVista_eval/evaluation/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e7a23ea3695f6437","mcp_get_code":{"code_sha256":"e7a23ea3695f6437"}},{"arxiv_id":"2409.14713","paper":"/paper/phantom-of-latent-for-large-language-and","title":"Phantom of Latent for Large Language and Vision Models","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"byungkwanlee/phantom","path":"eval/mathvista/calculate_score.py","file_url":"https://github.com/byungkwanlee/phantom/blob/HEAD/eval/mathvista/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e7a23ea3695f6437","mcp_get_code":{"code_sha256":"e7a23ea3695f6437"}},{"arxiv_id":"2409.00147","paper":"/paper/multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pengshuai-rin/multimath","path":"eval_mathvista/calculate_score.py","file_url":"https://github.com/pengshuai-rin/multimath/blob/HEAD/eval_mathvista/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e7a23ea3695f6437","mcp_get_code":{"code_sha256":"e7a23ea3695f6437"}},{"arxiv_id":"2405.15574","paper":"/paper/meteor-mamba-based-traversal-of-rationale-for","title":"Meteor: Mamba-based Traversal of Rationale for Large Language and Vision Models","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"byungkwanlee/meteor","path":"eval/mathvista/calculate_score.py","file_url":"https://github.com/byungkwanlee/meteor/blob/HEAD/eval/mathvista/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e7a23ea3695f6437","mcp_get_code":{"code_sha256":"e7a23ea3695f6437"}},{"arxiv_id":"2403.03003","paper":"/paper/feast-your-eyes-mixture-of-resolution","title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luogen1996/llava-hr","path":"llava_hr/eval/calculate_score.py","file_url":"https://github.com/luogen1996/llava-hr/blob/HEAD/llava_hr/eval/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e7a23ea3695f6437","mcp_get_code":{"code_sha256":"e7a23ea3695f6437"}},{"arxiv_id":"2402.10104","paper":"/paper/geoeval-benchmark-for-evaluating-llms-and","title":"GeoEval: Benchmark for Evaluating LLMs and Multi-Modal Models on Geometry Problem-Solving","date":"2024-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"geoeval/geoeval","path":"tool/caculate_score.py","file_url":"https://github.com/geoeval/geoeval/blob/HEAD/tool/caculate_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8b86d2b7f8a9bea3","mcp_get_code":{"code_sha256":"8b86d2b7f8a9bea3"}},{"arxiv_id":"2310.02255","paper":"/paper/mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lupantech/MathVista","path":"evaluation/calculate_score.py","file_url":"https://github.com/lupantech/MathVista/blob/HEAD/evaluation/calculate_score.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"CC-BY-SA-4.0","inline_ok":false,"code_sha256_prefix":"c367edb2cd4acf90","mcp_get_code":{"code_sha256":"c367edb2cd4acf90"}},{"arxiv_id":"Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","paper":null,"title":"arXiv:Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"guozix/DVLR","path":"eval/MathVerse/evaluation/calculate_score.py","file_url":"https://github.com/guozix/DVLR/blob/HEAD/eval/MathVerse/evaluation/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e7a23ea3695f6437","mcp_get_code":{"code_sha256":"e7a23ea3695f6437"}}]}