{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/parse-score","entry":"parse_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":49,"n_papers_ran":48,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":14,"n_samples_ran":13,"n_samples_fingerprinted":10,"n_places":51,"n_places_pointer_only":30,"by_status":{"ran_honours":1,"ran_violates":3,"ran_draft_wrong":2,"ran_fixture":0,"ran":7,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.07024","paper":"/paper/arxiv-2605-07024","title":"DELULU: A Verified Multi-Lingual Benchmark for Code Hallucination Detection in Fill-in-the-Middle Tasks","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"microsoft/delulu","path":"evaluations/run_delulu_judges.py","file_url":"https://github.com/microsoft/delulu/blob/HEAD/evaluations/run_delulu_judges.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"85a1174c53ba7c3a","mcp_get_code":{"code_sha256":"85a1174c53ba7c3a"}},{"arxiv_id":"2604.14463","paper":"/paper/arxiv-2604-14463","title":"Psychological Steering of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"kaistAI/FLASK","path":"gpt_review/gpt4_eval.py","file_url":"https://github.com/kaistAI/FLASK/blob/HEAD/gpt_review/gpt4_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"933335551c0e8a52","mcp_get_code":{"code_sha256":"933335551c0e8a52"}},{"arxiv_id":"2503.15621","paper":"/paper/llava-more-a-comparative-study-of-llms-and","title":"LLaVA-MORE: A Comparative Study of LLMs and Visual Backbones for Enhanced Visual Instruction Tuning","date":"2025-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aimagelab/LLaVA-MORE","path":"src/llava/eval/eval_gpt_review.py","file_url":"https://github.com/aimagelab/LLaVA-MORE/blob/HEAD/src/llava/eval/eval_gpt_review.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2503.00723","paper":"/paper/re-imagining-multimodal-instruction-tuning-a","title":"Re-Imagining Multimodal Instruction Tuning: A Representation View","date":"2025-03-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2412.02158","paper":"/paper/agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kki2eve/agri-llava","path":"agri_llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/kki2eve/agri-llava/blob/HEAD/agri_llava/eval/eval_gpt_review_visual.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2411.16725","paper":"/paper/textit-revelio-interpreting-and-leveraging","title":"$\\textit{Revelio}$: Interpreting and leveraging semantic information in diffusion models","date":"2024-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2411.00394","paper":"/paper/right-this-way-can-vlms-guide-us-to-see-more","title":"Right this way: Can VLMs Guide Us to See More to Answer Questions?","date":"2024-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2410.16198","paper":"/paper/improve-vision-language-model-chain-of","title":"Improve Vision Language Model Chain-of-thought Reasoning","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2410.03775","paper":"/paper/beyond-correlation-the-impact-of-human","title":"Beyond correlation: The Impact of Human Uncertainty in Measuring the Effectiveness of Automatic Evaluation and LLM-as-a-Judge","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/beyondcorrelation","path":"examples/judge_bench_example.py","file_url":"https://github.com/amazon-science/beyondcorrelation/blob/HEAD/examples/judge_bench_example.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0b9867bdf14b90fb","mcp_get_code":{"code_sha256":"0b9867bdf14b90fb"}},{"arxiv_id":"2410.02761","paper":"/paper/fakeshield-explainable-image-forgery","title":"FakeShield: Explainable Image Forgery Detection and Localization via Multi-modal Large Language Models","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2409.19951","paper":"/paper/law-of-the-weakest-link-cross-capabilities-of","title":"Law of the Weakest Link: Cross Capabilities of Large Language Models","date":"2024-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/llm-cross-capabilities","path":"evaluation/evaluate_response.py","file_url":"https://github.com/facebookresearch/llm-cross-capabilities/blob/HEAD/evaluation/evaluate_response.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"f73ce4685c268b2d","mcp_get_code":{"code_sha256":"f73ce4685c268b2d"}},{"arxiv_id":"2408.15545","paper":"/paper/scilitllm-how-to-adapt-llms-for-scientific","title":"SciLitLLM: How to Adapt LLMs for Scientific Literature Understanding","date":"2024-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dptech-corp/Uni-SMART","path":"SciLitLLM/cpt/quality_control/llama_infer.py","file_url":"https://github.com/dptech-corp/Uni-SMART/blob/HEAD/SciLitLLM/cpt/quality_control/llama_infer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"978a0952e0e705b2","mcp_get_code":{"code_sha256":"978a0952e0e705b2"}},{"arxiv_id":"2408.13906","paper":"/paper/convis-contrastive-decoding-with","title":"ConVis: Contrastive Decoding with Hallucination Visualization for Mitigating Hallucinations in Multimodal Large Language Models","date":"2024-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yejipark-m/convis","path":"eval/eval_gpt_review_bench.py","file_url":"https://github.com/yejipark-m/convis/blob/HEAD/eval/eval_gpt_review_bench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2408.01417","paper":"/paper/2408-01417","title":"Talk Less, Interact Better: Evaluating In-context Conversational Adaptation in Multimodal LLMs","date":"2024-08-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2407.16198","paper":"/paper/inf-llava-dual-perspective-perception-for","title":"INF-LLaVA: Dual-perspective Perception for High-Resolution Multimodal Large Language Model","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2406.11833","paper":"/paper/mmdu-a-multi-turn-multi-image-dialog","title":"MMDU: A Multi-Turn Multi-Image Dialog Understanding Benchmark and Instruction-Tuning Dataset for LVLMs","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2406.11823","paper":"/paper/on-efficient-language-and-vision-assistants","title":"On Efficient Language and Vision Assistants for Visually-Situated Natural Language Understanding: What Matters in Reading and Reasoning","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/elva","path":"Elva/eval_gpt_review_parsing_bench.py","file_url":"https://github.com/naver-ai/elva/blob/HEAD/Elva/eval_gpt_review_parsing_bench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2406.04371","paper":"/paper/phased-instruction-fine-tuning-for-large","title":"Phased Instruction Fine-Tuning for Large Language Models","date":"2024-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xubuvd/phasedsft","path":"evaluation/win_tie_loss_stat.py","file_url":"https://github.com/xubuvd/phasedsft/blob/HEAD/evaluation/win_tie_loss_stat.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d9b9b1f09f63f865","mcp_get_code":{"code_sha256":"d9b9b1f09f63f865"}},{"arxiv_id":"2405.18842","paper":"/paper/descriptive-image-quality-assessment-in-the","title":"Descriptive Image Quality Assessment in the Wild","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XPixelGroup/DepictQA","path":"src/eval/cal_gpt4_score_detail_v1.py","file_url":"https://github.com/XPixelGroup/DepictQA/blob/HEAD/src/eval/cal_gpt4_score_detail_v1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2405.18842","paper":"/paper/descriptive-image-quality-assessment-in-the","title":"Descriptive Image Quality Assessment in the Wild","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XPixelGroup/DepictQA","path":"src/eval/cal_gpt4_score_detail_v2.py","file_url":"https://github.com/XPixelGroup/DepictQA/blob/HEAD/src/eval/cal_gpt4_score_detail_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d22ddef4b1ae67fd","mcp_get_code":{"code_sha256":"d22ddef4b1ae67fd"}},{"arxiv_id":"2405.17220","paper":"/paper/rlaif-v-aligning-mllms-through-open-source-ai","title":"RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rlhf-v/rlhf-v","path":"eval/eval_gpt_review_llava.py","file_url":"https://github.com/rlhf-v/rlhf-v/blob/HEAD/eval/eval_gpt_review_llava.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2405.14974","paper":"/paper/lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2404.13046","paper":"/paper/mova-adapting-mixture-of-vision-experts-to","title":"MoVA: Adapting Mixture of Vision Experts to Multimodal Context","date":"2024-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TempleX98/MoVA","path":"mova/eval/eval_gpt_review.py","file_url":"https://github.com/TempleX98/MoVA/blob/HEAD/mova/eval/eval_gpt_review.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2404.13013","paper":"/paper/groma-localized-visual-tokenization-for","title":"Groma: Localized Visual Tokenization for Grounding Multimodal Large Language Models","date":"2024-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FoundationVision/Groma","path":"groma/eval/eval_gpt_review_visual.py","file_url":"https://github.com/FoundationVision/Groma/blob/HEAD/groma/eval/eval_gpt_review_visual.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2403.18252","paper":"/paper/beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/visual-table","path":"llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/lavi-lab/visual-table/blob/HEAD/llava/eval/eval_gpt_review_visual.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2403.16999","paper":"/paper/visual-cot-unleashing-chain-of-thought","title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","date":"2024-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepcs233/visual-cot","path":"llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/deepcs233/visual-cot/blob/HEAD/llava/eval/eval_gpt_review_visual.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2403.09792","paper":"/paper/images-are-achilles-heel-of-alignment","title":"Images are Achilles' Heel of Alignment: Exploiting Visual Vulnerabilities for Jailbreaking Multimodal Large Language Models","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2402.13561","paper":"/paper/cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hitsz-tmg/cognitive-visual-language-mapper","path":"LLaVA/llava/eval/eval_gpt_review_visual.py","file_url":"https://github.com/hitsz-tmg/cognitive-visual-language-mapper/blob/HEAD/LLaVA/llava/eval/eval_gpt_review_visual.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2402.13093","paper":"/paper/event-level-knowledge-editing","title":"Event-level Knowledge Editing","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-keg/event-level-knowledge-editing","path":"examiner/parse_result.py","file_url":"https://github.com/thu-keg/event-level-knowledge-editing/blob/HEAD/examiner/parse_result.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"9e0abb69457fe1bc","mcp_get_code":{"code_sha256":"9e0abb69457fe1bc"}},{"arxiv_id":"2402.13093","paper":"/paper/event-level-knowledge-editing","title":"Event-level Knowledge Editing","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-keg/event-level-knowledge-editing","path":"examiner/parse_result_locality.py","file_url":"https://github.com/thu-keg/event-level-knowledge-editing/blob/HEAD/examiner/parse_result_locality.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"7da1f09a1752b039","mcp_get_code":{"code_sha256":"7da1f09a1752b039"}},{"arxiv_id":"2402.10884","paper":"/paper/multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2402.07319","paper":"/paper/odin-disentangled-reward-mitigates-hacking-in","title":"ODIN: Disentangled Reward Mitigates Hacking in RLHF","date":"2024-02-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2402.04833","paper":"/paper/long-is-more-for-alignment-a-simple-but-tough","title":"Long Is More for Alignment: A Simple but Tough-to-Beat Baseline for Instruction Fine-Tuning","date":"2024-02-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2312.10032","paper":"/paper/osprey-pixel-understanding-with-visual","title":"Osprey: Pixel Understanding with Visual Instruction Tuning","date":"2023-12-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2312.09366","paper":"/paper/arabic-mini-climategpt-a-climate-change-and","title":"Arabic Mini-ClimateGPT : A Climate Change and Sustainability Tailored Arabic LLM","date":"2023-12-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2312.06968","paper":"/paper/hallucination-augmented-contrastive-learning","title":"Hallucination Augmented Contrastive Learning for Multimodal Large Language Model","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2312.04746","paper":"/paper/quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aldraus/quilt-llava","path":"llava/eval/quilt_gpt_eval.py","file_url":"https://github.com/aldraus/quilt-llava/blob/HEAD/llava/eval/quilt_gpt_eval.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2312.03818","paper":"/paper/alpha-clip-a-clip-model-focusing-on-wherever","title":"Alpha-CLIP: A CLIP Model Focusing on Wherever You Want","date":"2023-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2312.00374","paper":"/paper/unleashing-cheapfakes-through-trojan-plugins","title":"The Philosopher's Stone: Trojaning Plugins of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2023-12","title_source":"archive","repo":"chichidd/llm-lora-trojan","path":"eval/gpt_review.py","file_url":"https://github.com/chichidd/llm-lora-trojan/blob/HEAD/eval/gpt_review.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2310.03744","paper":"/paper/improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2309.09958","paper":"/paper/an-empirical-study-of-scaling-instruct-tuned","title":"An Empirical Study of Scaling Instruct-Tuned Large Multimodal Models","date":"2023-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"763f6fe20bd52fbb","mcp_get_code":{"code_sha256":"763f6fe20bd52fbb"}},{"arxiv_id":"2308.03279","paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"universal-ner/universal-ner","path":"src/train/fastchat/eval/eval_gpt_review.py","file_url":"https://github.com/universal-ner/universal-ner/blob/HEAD/src/train/fastchat/eval/eval_gpt_review.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2307.10928","paper":"/paper/flask-fine-grained-language-model-evaluation","title":"FLASK: Fine-grained Language Model Evaluation based on Alignment Skill Sets","date":"2023-07-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaistai/flask","path":"gpt_review/gpt4_eval.py","file_url":"https://github.com/kaistai/flask/blob/HEAD/gpt_review/gpt4_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"933335551c0e8a52","mcp_get_code":{"code_sha256":"933335551c0e8a52"}},{"arxiv_id":"2307.08701","paper":"/paper/alpagasus-training-a-better-alpaca-with-fewer","title":"AlpaGasus: Training A Better Alpaca with Fewer Data","date":"2023-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gpt4life/alpagasus","path":"rating/filter.py","file_url":"https://github.com/gpt4life/alpagasus/blob/HEAD/rating/filter.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b748891b4b063b83","mcp_get_code":{"code_sha256":"b748891b4b063b83"}},{"arxiv_id":"2307.02762","paper":"/paper/prd-peer-rank-and-discussion-improve-large","title":"PRD: Peer Rank and Discussion Improve Large Language Model based Evaluations","date":"2023-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcdnlp/prd","path":"peer_rank/eval_bard_review.py","file_url":"https://github.com/bcdnlp/prd/blob/HEAD/peer_rank/eval_bard_review.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2305.14314","paper":"/paper/qlora-efficient-finetuning-of-quantized-llms","title":"QLoRA: Efficient Finetuning of Quantized LLMs","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"artidoro/qlora","path":"eval/eval_gpt_review.py","file_url":"https://github.com/artidoro/qlora/blob/HEAD/eval/eval_gpt_review.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2305.12870","paper":"/paper/lion-adversarial-distillation-of-closed","title":"Lion: Adversarial Distillation of Proprietary Large Language Models","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yjiangcm/lion","path":"src/chatgpt_referee.py","file_url":"https://github.com/yjiangcm/lion/blob/HEAD/src/chatgpt_referee.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"eb6629df9a940cb4","mcp_get_code":{"code_sha256":"eb6629df9a940cb4"}},{"arxiv_id":"2305.12031","paper":"/paper/clinical-camel-an-open-source-expert-level","title":"Clinical Camel: An Open Expert-Level Medical Language Model with Dialogue-Based Knowledge Encoding","date":"2023-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bowang-lab/clinical-camel","path":"evaluation/eval_gpt_review.py","file_url":"https://github.com/bowang-lab/clinical-camel/blob/HEAD/evaluation/eval_gpt_review.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"8049b382893c73dd","mcp_get_code":{"code_sha256":"8049b382893c73dd"}},{"arxiv_id":"2303.16634","paper":"/paper/gpteval-nlg-evaluation-using-gpt-4-with","title":"G-Eval: NLG Evaluation using GPT-4 with Better Human Alignment","date":"2023-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"megagonlabs/llm-longeval","path":"eval/evaluator.py","file_url":"https://github.com/megagonlabs/llm-longeval/blob/HEAD/eval/evaluator.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"f662bdaf543ab06e","mcp_get_code":{"code_sha256":"f662bdaf543ab06e"}},{"arxiv_id":"2303.08774","paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"b748891b4b063b83","mcp_get_code":{"code_sha256":"b748891b4b063b83"}},{"arxiv_id":"2024.findings-acl.341","paper":null,"title":"arXiv:2024.findings-acl.341","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"xubuvd/PhasedSFT","path":"evaluation/win_tie_loss_stat.py","file_url":"https://github.com/xubuvd/PhasedSFT/blob/HEAD/evaluation/win_tie_loss_stat.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d9b9b1f09f63f865","mcp_get_code":{"code_sha256":"d9b9b1f09f63f865"}}]}