{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-eval","entry":"get_eval","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":18,"n_papers_ran":9,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":21,"n_samples_ran":9,"n_samples_fingerprinted":2,"n_places":24,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":9,"unverified":12},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2503.15621","paper":"/paper/llava-more-a-comparative-study-of-llms-and","title":"LLaVA-MORE: A Comparative Study of LLMs and Visual Backbones for Enhanced Visual Instruction Tuning","date":"2025-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aimagelab/LLaVA-MORE","path":"src/llava/eval/eval_gpt_review_bench.py","file_url":"https://github.com/aimagelab/LLaVA-MORE/blob/HEAD/src/llava/eval/eval_gpt_review_bench.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fb48214e8ea84c17","mcp_get_code":{"code_sha256":"fb48214e8ea84c17"}},{"arxiv_id":"2503.15621","paper":"/paper/llava-more-a-comparative-study-of-llms-and","title":"LLaVA-MORE: A Comparative Study of LLMs and Visual Backbones for Enhanced Visual Instruction Tuning","date":"2025-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aimagelab/LLaVA-MORE","path":"src/llava/eval/eval_gpt_review.py","file_url":"https://github.com/aimagelab/LLaVA-MORE/blob/HEAD/src/llava/eval/eval_gpt_review.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"64a3ae0bc1b51280","mcp_get_code":{"code_sha256":"64a3ae0bc1b51280"}},{"arxiv_id":"2412.02158","paper":"/paper/agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kki2eve/agri-llava","path":"agri_llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/kki2eve/agri-llava/blob/HEAD/agri_llava/eval/eval_gpt_review_visual.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5f955676f48705c9","mcp_get_code":{"code_sha256":"5f955676f48705c9"}},{"arxiv_id":"2408.13906","paper":"/paper/convis-contrastive-decoding-with","title":"ConVis: Contrastive Decoding with Hallucination Visualization for Mitigating Hallucinations in Multimodal Large Language Models","date":"2024-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yejipark-m/convis","path":"eval/eval_gpt_review_bench.py","file_url":"https://github.com/yejipark-m/convis/blob/HEAD/eval/eval_gpt_review_bench.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"13cb3dd1e9e1ed71","mcp_get_code":{"code_sha256":"13cb3dd1e9e1ed71"}},{"arxiv_id":"2406.11823","paper":"/paper/on-efficient-language-and-vision-assistants","title":"On Efficient Language and Vision Assistants for Visually-Situated Natural Language Understanding: What Matters in Reading and Reasoning","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/elva","path":"Elva/eval_gpt_review_parsing_bench.py","file_url":"https://github.com/naver-ai/elva/blob/HEAD/Elva/eval_gpt_review_parsing_bench.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a5b1f40fc357a806","mcp_get_code":{"code_sha256":"a5b1f40fc357a806"}},{"arxiv_id":"2405.20690","paper":"/paper/unleashing-the-potential-of-diffusion-models","title":"Unleashing the Potential of Diffusion Models for Incomplete Data Imputation","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hengruizhang98/DiffPuter","path":"dataset.py","file_url":"https://github.com/hengruizhang98/DiffPuter/blob/HEAD/dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2f212d29fbe5460e","mcp_get_code":{"code_sha256":"2f212d29fbe5460e"}},{"arxiv_id":"2405.20690","paper":"/paper/unleashing-the-potential-of-diffusion-models","title":"Unleashing the Potential of Diffusion Models for Incomplete Data Imputation","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hengruizhang98/DiffPuter","path":"baselines/data_utils.py","file_url":"https://github.com/hengruizhang98/DiffPuter/blob/HEAD/baselines/data_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b1c68b94d50528a2","mcp_get_code":{"code_sha256":"b1c68b94d50528a2"}},{"arxiv_id":"2405.17220","paper":"/paper/rlaif-v-aligning-mllms-through-open-source-ai","title":"RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rlhf-v/rlhf-v","path":"eval/gpt4_grpc.py","file_url":"https://github.com/rlhf-v/rlhf-v/blob/HEAD/eval/gpt4_grpc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"77010a13f8bd8d11","mcp_get_code":{"code_sha256":"77010a13f8bd8d11"}},{"arxiv_id":"2404.13013","paper":"/paper/groma-localized-visual-tokenization-for","title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","date":"2024-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FoundationVision/Groma","path":"groma/eval/eval_gpt_review_visual.py","file_url":"https://github.com/FoundationVision/Groma/blob/HEAD/groma/eval/eval_gpt_review_visual.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55b3991e52737846","mcp_get_code":{"code_sha256":"55b3991e52737846"}},{"arxiv_id":"2403.18252","paper":"/paper/beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/visual-table","path":"llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/lavi-lab/visual-table/blob/HEAD/llava/eval/eval_gpt_review_visual.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fb48214e8ea84c17","mcp_get_code":{"code_sha256":"fb48214e8ea84c17"}},{"arxiv_id":"2403.16999","paper":"/paper/visual-cot-unleashing-chain-of-thought","title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","date":"2024-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepcs233/visual-cot","path":"llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/deepcs233/visual-cot/blob/HEAD/llava/eval/eval_gpt_review_visual.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fb48214e8ea84c17","mcp_get_code":{"code_sha256":"fb48214e8ea84c17"}},{"arxiv_id":"2402.13561","paper":"/paper/cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hitsz-tmg/cognitive-visual-language-mapper","path":"LLaVA/llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/hitsz-tmg/cognitive-visual-language-mapper/blob/HEAD/LLaVA/llava/eval/eval_gpt_review_visual.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fb48214e8ea84c17","mcp_get_code":{"code_sha256":"fb48214e8ea84c17"}},{"arxiv_id":"2312.04746","paper":"/paper/quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aldraus/quilt-llava","path":"llava/eval/quilt_gpt_eval.py","file_url":"https://github.com/aldraus/quilt-llava/blob/HEAD/llava/eval/quilt_gpt_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c00535a655c69753","mcp_get_code":{"code_sha256":"c00535a655c69753"}},{"arxiv_id":"2312.00374","paper":"/paper/unleashing-cheapfakes-through-trojan-plugins","title":"The Philosopher's Stone: Trojaning Plugins of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2023-12","title_source":"archive","repo":"chichidd/llm-lora-trojan","path":"eval/gpt_review.py","file_url":"https://github.com/chichidd/llm-lora-trojan/blob/HEAD/eval/gpt_review.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"191943d87af4d1bb","mcp_get_code":{"code_sha256":"191943d87af4d1bb"}},{"arxiv_id":"2310.20410","paper":"/paper/followbench-a-multi-level-fine-grained","title":"FollowBench: A Multi-level Fine-grained Constraints Following Benchmark for Large Language Models","date":"2023-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yjiangcm/followbench","path":"code/llm_eval.py","file_url":"https://github.com/yjiangcm/followbench/blob/HEAD/code/llm_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4cd97990a8cad73f","mcp_get_code":{"code_sha256":"4cd97990a8cad73f"}},{"arxiv_id":"2310.20410","paper":"/paper/followbench-a-multi-level-fine-grained","title":"FollowBench: A Multi-level Fine-grained Constraints Following Benchmark for Large Language Models","date":"2023-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yjiangcm/followbench","path":"code_zh/llm_eval.py","file_url":"https://github.com/yjiangcm/followbench/blob/HEAD/code_zh/llm_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"570dd47754890fe2","mcp_get_code":{"code_sha256":"570dd47754890fe2"}},{"arxiv_id":"2310.01377","paper":"/paper/ultrafeedback-boosting-language-models-with","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/ultrafeedback","path":"src/data_annotation/annotate_critique.py","file_url":"https://github.com/thunlp/ultrafeedback/blob/HEAD/src/data_annotation/annotate_critique.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a506738814a38c14","mcp_get_code":{"code_sha256":"a506738814a38c14"}},{"arxiv_id":"2310.01377","paper":"/paper/ultrafeedback-boosting-language-models-with","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/ultrafeedback","path":"src/data_annotation/fix_overall_score_issue.py","file_url":"https://github.com/thunlp/ultrafeedback/blob/HEAD/src/data_annotation/fix_overall_score_issue.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9f8a0e28964db553","mcp_get_code":{"code_sha256":"9f8a0e28964db553"}},{"arxiv_id":"2310.01377","paper":"/paper/ultrafeedback-boosting-language-models-with","title":"UltraFeedback: Boosting Language Models with Scaled AI Feedback","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/ultrafeedback","path":"src/data_annotation/annotate_preference.py","file_url":"https://github.com/thunlp/ultrafeedback/blob/HEAD/src/data_annotation/annotate_preference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"38d99fe231928f20","mcp_get_code":{"code_sha256":"38d99fe231928f20"}},{"arxiv_id":"2308.03279","paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"universal-ner/universal-ner","path":"src/train/fastchat/eval/eval_gpt_review.py","file_url":"https://github.com/universal-ner/universal-ner/blob/HEAD/src/train/fastchat/eval/eval_gpt_review.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"26ac62acef82f73c","mcp_get_code":{"code_sha256":"26ac62acef82f73c"}},{"arxiv_id":"2307.02762","paper":"/paper/prd-peer-rank-and-discussion-improve-large","title":"PRD: Peer Rank and Discussion Improve Large Language Model based Evaluations","date":"2023-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcdnlp/prd","path":"peer_rank/eval_bard_review.py","file_url":"https://github.com/bcdnlp/prd/blob/HEAD/peer_rank/eval_bard_review.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f296a758b3b2f6d2","mcp_get_code":{"code_sha256":"f296a758b3b2f6d2"}},{"arxiv_id":"2305.12870","paper":"/paper/lion-adversarial-distillation-of-closed","title":"Lion: Adversarial Distillation of Proprietary Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yjiangcm/lion","path":"src/chatgpt_inference.py","file_url":"https://github.com/yjiangcm/lion/blob/HEAD/src/chatgpt_inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8cd557465a1f5553","mcp_get_code":{"code_sha256":"8cd557465a1f5553"}},{"arxiv_id":"2305.12870","paper":"/paper/lion-adversarial-distillation-of-closed","title":"Lion: Adversarial Distillation of Proprietary Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yjiangcm/lion","path":"src/chatgpt_referee.py","file_url":"https://github.com/yjiangcm/lion/blob/HEAD/src/chatgpt_referee.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5823dba3817cc067","mcp_get_code":{"code_sha256":"5823dba3817cc067"}},{"arxiv_id":"2305.01122","paper":"/paper/learning-controllable-adaptive-simulation-for","title":"Learning Controllable Adaptive Simulation for Multi-resolution Physics","date":"2023-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"snap-stanford/lamp","path":"analysis_2d_full.py","file_url":"https://github.com/snap-stanford/lamp/blob/HEAD/analysis_2d_full.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32a2b6aa244f4c8c","mcp_get_code":{"code_sha256":"32a2b6aa244f4c8c"}}]}