{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-jsonl","entry":"load_jsonl","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":179,"n_papers_ran":107,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":175,"n_samples_ran":97,"n_samples_fingerprinted":4,"n_places":210,"n_places_pointer_only":79,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":22,"ran_fixture":0,"ran":75,"unverified":78},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.03370","paper":"/paper/arxiv-2609-03370","title":"FrameBench:A Language Understanding Benchmark Based on Frame Semantics","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"SasanoLab/FrameBench","path":"src/generation/utils/utils.py","file_url":"https://github.com/SasanoLab/FrameBench/blob/HEAD/src/generation/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"22a081a82f22f469","mcp_get_code":{"code_sha256":"22a081a82f22f469"}},{"arxiv_id":"2609.01556","paper":"/paper/arxiv-2609-01556","title":"Retrieved but not ranked: surface-form bias in structural retrieval, from mathematics to agent trajectories","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"nabirarashid/structural-retrieval","path":"src/data.py","file_url":"https://github.com/nabirarashid/structural-retrieval/blob/HEAD/src/data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"88655287e149326b","mcp_get_code":{"code_sha256":"88655287e149326b"}},{"arxiv_id":"2608.30156","paper":"/paper/arxiv-2608-30156","title":"Reactivating Test-Time Scaling for Plane Geometry Problem Solving","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Jason8Kang/ReTTS-PGPS","path":"src/retts_pgp/eval/evaluate.py","file_url":"https://github.com/Jason8Kang/ReTTS-PGPS/blob/HEAD/src/retts_pgp/eval/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"5782d82a3417e0a7","mcp_get_code":{"code_sha256":"5782d82a3417e0a7"}},{"arxiv_id":"2608.19181","paper":"/paper/arxiv-2608-19181","title":"Beyond Teacher Likelihood: Group-Calibrated On-Policy Distillation for Long-Context Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"SolereZhang/GC-OPD","path":"evaluation/aggregate_main_table.py","file_url":"https://github.com/SolereZhang/GC-OPD/blob/HEAD/evaluation/aggregate_main_table.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5627227cf9041a8d","mcp_get_code":{"code_sha256":"5627227cf9041a8d"}},{"arxiv_id":"2608.11755","paper":"/paper/arxiv-2608-11755","title":"MuseCritic: Learning Multi-Aspect Song Rewards through Natural-Language Aesthetic Critiques","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"WuqnEl/MuseCritic","path":"eval/compute_corr.py","file_url":"https://github.com/WuqnEl/MuseCritic/blob/HEAD/eval/compute_corr.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"90450e4180be9ae8","mcp_get_code":{"code_sha256":"90450e4180be9ae8"}},{"arxiv_id":"2608.10008","paper":"/paper/arxiv-2608-10008","title":"Do LLM Recommenders Know When They're Hallucinating? Auditing Confidence Calibration in Catalog Faithfulness","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"rsrijith/cikm26-catalog-faithfulness","path":"code/prep_yelp.py","file_url":"https://github.com/rsrijith/cikm26-catalog-faithfulness/blob/HEAD/code/prep_yelp.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"7a1382f2ea061f72","mcp_get_code":{"code_sha256":"7a1382f2ea061f72"}},{"arxiv_id":"2608.08557","paper":"/paper/arxiv-2608-08557","title":"OpenVisTool: An Open Recipe for Synthesizing Instructive Visual Tool-Use Trajectories","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Changhao-Xiang/OpenVisTool","path":"distill/filter/evaluate_instructive_trajectory_ablation.py","file_url":"https://github.com/Changhao-Xiang/OpenVisTool/blob/HEAD/distill/filter/evaluate_instructive_trajectory_ablation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fd449ff795ee3154","mcp_get_code":{"code_sha256":"fd449ff795ee3154"}},{"arxiv_id":"2608.08557","paper":"/paper/arxiv-2608-08557","title":"OpenVisTool: An Open Recipe for Synthesizing Instructive Visual Tool-Use Trajectories","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Changhao-Xiang/OpenVisTool","path":"distill/filter/filter_tool_gain_with_prefix.py","file_url":"https://github.com/Changhao-Xiang/OpenVisTool/blob/HEAD/distill/filter/filter_tool_gain_with_prefix.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fa52ba8cf457edcf","mcp_get_code":{"code_sha256":"fa52ba8cf457edcf"}},{"arxiv_id":"2608.03467","paper":"/paper/arxiv-2608-03467","title":"When Correct Solutions Repeat: Rarity-Aware Credit Redistribution for GRPO","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO","path":"analyze_mechanism.py","file_url":"https://github.com/CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO/blob/HEAD/analyze_mechanism.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f835ffe0b6606634","mcp_get_code":{"code_sha256":"f835ffe0b6606634"}},{"arxiv_id":"2608.03467","paper":"/paper/arxiv-2608-03467","title":"When Correct Solutions Repeat: Rarity-Aware Credit Redistribution for GRPO","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO","path":"eval_full_clusters.py","file_url":"https://github.com/CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO/blob/HEAD/eval_full_clusters.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"40d46e8d30cf8006","mcp_get_code":{"code_sha256":"40d46e8d30cf8006"}},{"arxiv_id":"2608.03467","paper":"/paper/arxiv-2608-03467","title":"When Correct Solutions Repeat: Rarity-Aware Credit Redistribution for GRPO","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO","path":"eval_vllm.py","file_url":"https://github.com/CzZ12/When-Correct-Solutions-Repeat-Rarity-Aware-Credit-Redistribution-for-GRPO/blob/HEAD/eval_vllm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"70091d81e0d906bd","mcp_get_code":{"code_sha256":"70091d81e0d906bd"}},{"arxiv_id":"2608.03063","paper":"/paper/arxiv-2608-03063","title":"SeqLLM: Augmenting LLMs with Behavioral-Sequence Modeling for High-Stakes Decisions at WeChat Pay","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"125jx/SeqLLM","path":"evaluation/eval_item_understanding_task/infer_recprobe_vllm.py","file_url":"https://github.com/125jx/SeqLLM/blob/HEAD/evaluation/eval_item_understanding_task/infer_recprobe_vllm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ffd4c6463dd0a477","mcp_get_code":{"code_sha256":"ffd4c6463dd0a477"}},{"arxiv_id":"2608.01358","paper":"/paper/arxiv-2608-01358","title":"HopRefusalBench: Diagnosing Refusal Failures in Search-Augmented Agents for Multi-Hop Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"JiananXie/HopRefusalBench","path":"evaluate.py","file_url":"https://github.com/JiananXie/HopRefusalBench/blob/HEAD/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2820cd4e6747f653","mcp_get_code":{"code_sha256":"2820cd4e6747f653"}},{"arxiv_id":"2607.28439","paper":"/paper/arxiv-2607-28439","title":"Beyond a Single Judge: The Evidence-Grounded, Social-Weighted Persona Panel for Generative UI Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"Wuzheng02/ESPP","path":"pqa_construction/common.py","file_url":"https://github.com/Wuzheng02/ESPP/blob/HEAD/pqa_construction/common.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b1828f7666144a8","mcp_get_code":{"code_sha256":"0b1828f7666144a8"}},{"arxiv_id":"2607.28439","paper":"/paper/arxiv-2607-28439","title":"Beyond a Single Judge: The Evidence-Grounded, Social-Weighted Persona Panel for Generative UI Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"Wuzheng02/ESPP","path":"scoring_pipeline/common.py","file_url":"https://github.com/Wuzheng02/ESPP/blob/HEAD/scoring_pipeline/common.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ecfe93b1c22b03a1","mcp_get_code":{"code_sha256":"ecfe93b1c22b03a1"}},{"arxiv_id":"2607.14895","paper":"/paper/arxiv-2607-14895","title":"Leveraging Instruction Tuning and Merging for Reasoning Model Adaptation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"eth-sri/rlm-training-merging","path":"src/evaluation/text_summarization.py","file_url":"https://github.com/eth-sri/rlm-training-merging/blob/HEAD/src/evaluation/text_summarization.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"455c557e67fba06c","mcp_get_code":{"code_sha256":"455c557e67fba06c"}},{"arxiv_id":"2607.01022","paper":"/paper/arxiv-2607-01022","title":"A Unified Benchmarking Framework for Spatiotemporal Event Modeling","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"YahyaAalaila/seahorse","path":"seahorse/utils.py","file_url":"https://github.com/YahyaAalaila/seahorse/blob/HEAD/seahorse/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fd2e5730d6853aae","mcp_get_code":{"code_sha256":"fd2e5730d6853aae"}},{"arxiv_id":"2606.28572","paper":"/paper/arxiv-2606-28572","title":"Geometric Measurements of the Axiom of Choice in Neural Proof Embeddings","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"rodrgo/geometric-axiom-of-choice","path":"experiments/classical_ablation_analyze.py","file_url":"https://github.com/rodrgo/geometric-axiom-of-choice/blob/HEAD/experiments/classical_ablation_analyze.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"83d5da6867b97795","mcp_get_code":{"code_sha256":"83d5da6867b97795"}},{"arxiv_id":"2606.26979","paper":"/paper/arxiv-2606-26979","title":"How Much Static Structure Do Code Agents Need? A Study of Deterministic Anchoring","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"mathieu0905/Code-Anchor","path":"evaluation/eval_metric.py","file_url":"https://github.com/mathieu0905/Code-Anchor/blob/HEAD/evaluation/eval_metric.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"846475125bc11ceb","mcp_get_code":{"code_sha256":"846475125bc11ceb"}},{"arxiv_id":"2606.26474","paper":"/paper/arxiv-2606-26474","title":"Localizing RL-Induced Tool Use to a Single Crosscoder Feature","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Antebe/model_diffing_crosscoders","path":"analysis/run_xcoder_hparams_analysis.py","file_url":"https://github.com/Antebe/model_diffing_crosscoders/blob/HEAD/analysis/run_xcoder_hparams_analysis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b7bc84fc9ab32ca0","mcp_get_code":{"code_sha256":"b7bc84fc9ab32ca0"}},{"arxiv_id":"2606.26130","paper":"/paper/arxiv-2606-26130","title":"Thinking Like a Scientist? A Structural Study of LLM-Generated Research Methods","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"francescacarlon/Thinking-Like-a-Scientist","path":"src/pipeline/compare_model_swap.py","file_url":"https://github.com/francescacarlon/Thinking-Like-a-Scientist/blob/HEAD/src/pipeline/compare_model_swap.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"041e02ba2162b2a0","mcp_get_code":{"code_sha256":"041e02ba2162b2a0"}},{"arxiv_id":"2606.18781","paper":"/paper/arxiv-2606-18781","title":"Lost in a Single Vector: Improving Long-Document Retrieval with Chunk Evidence Aggregation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"PunchlineAAAA/DICE","path":"ReasonAug/bright_verify.py","file_url":"https://github.com/PunchlineAAAA/DICE/blob/HEAD/ReasonAug/bright_verify.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"58e81c79c7830cc9","mcp_get_code":{"code_sha256":"58e81c79c7830cc9"}},{"arxiv_id":"2606.15307","paper":"/paper/arxiv-2606-15307","title":"Adapting Reinforcement Learning with Chain-of-Thought Supervision for Explainable Detection of Hateful and Propagandistic Memes","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"MohamedBayan/MemeReason","path":"annotation/azure_batch.py","file_url":"https://github.com/MohamedBayan/MemeReason/blob/HEAD/annotation/azure_batch.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bcc2eb61ab57d401","mcp_get_code":{"code_sha256":"bcc2eb61ab57d401"}},{"arxiv_id":"2606.13802","paper":"/paper/arxiv-2606-13802","title":"A Benchmark and Framework for Evaluating Next Action Predictions in Spreadsheets","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Tej-55/NAPE","path":"finetuning/data_preparation.py","file_url":"https://github.com/Tej-55/NAPE/blob/HEAD/finetuning/data_preparation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"cf97dcd92963847e","mcp_get_code":{"code_sha256":"cf97dcd92963847e"}},{"arxiv_id":"2606.12789","paper":"/paper/arxiv-2606-12789","title":"How Fine-Grained Should a RAG Benchmark Be? A Hierarchical Framework for Synthetic Question Generation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"fensorechase/rag-diverse-benchmarks-synthetic-qa","path":"rag_system/generate/complete_analysis.py","file_url":"https://github.com/fensorechase/rag-diverse-benchmarks-synthetic-qa/blob/HEAD/rag_system/generate/complete_analysis.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e5a2f31357586772","mcp_get_code":{"code_sha256":"e5a2f31357586772"}},{"arxiv_id":"2606.10528","paper":"/paper/arxiv-2606-10528","title":"Representation-Aware Advantage Estimation: Your Reward Model Provides More Than A Scalar Output","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"lmarena/arena-hard-auto","path":"qa_browser.py","file_url":"https://github.com/lmarena/arena-hard-auto/blob/HEAD/qa_browser.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d049c92b5f5adfa3","mcp_get_code":{"code_sha256":"d049c92b5f5adfa3"}},{"arxiv_id":"2606.08969","paper":"/paper/arxiv-2606-08969","title":"CARE: A Conformal Safety Layer for Medical Summarization","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"som-shahlab/CARE","path":"care/utils.py","file_url":"https://github.com/som-shahlab/CARE/blob/HEAD/care/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d062537281fbcda4","mcp_get_code":{"code_sha256":"d062537281fbcda4"}},{"arxiv_id":"2606.05478","paper":"/paper/arxiv-2606-05478","title":"Can We Predict The Human Preference For Text-to-Image Content Prior To Generation And Is It Even Useful To Do So?","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"LSU-ATHENA/HPM-Predict","path":"gen_dataset/run_hunyuan_rank100_extension.py","file_url":"https://github.com/LSU-ATHENA/HPM-Predict/blob/HEAD/gen_dataset/run_hunyuan_rank100_extension.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cfd35648dca1b0b7","mcp_get_code":{"code_sha256":"cfd35648dca1b0b7"}},{"arxiv_id":"2606.03980","paper":"/paper/arxiv-2606-03980","title":"Skill-RM: Unifying Heterogeneous Evaluation Criteria via Agent Skill","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Qwen-Applications/Skill-RM","path":"experiments/if_rewardbench/io_utils.py","file_url":"https://github.com/Qwen-Applications/Skill-RM/blob/HEAD/experiments/if_rewardbench/io_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6f16fa558d7424a4","mcp_get_code":{"code_sha256":"6f16fa558d7424a4"}},{"arxiv_id":"2606.03391","paper":"/paper/arxiv-2606-03391","title":"When Model Merging Breaks Routing: Training-Free Calibration for MoE","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"huangcb01/HARC","path":"src/utils.py","file_url":"https://github.com/huangcb01/HARC/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c3dd1094f3b76096","mcp_get_code":{"code_sha256":"c3dd1094f3b76096"}},{"arxiv_id":"2606.03391","paper":"/paper/arxiv-2606-03391","title":"When Model Merging Breaks Routing: Training-Free Calibration for MoE","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"huangcb01/HARC","path":"src/merge_method/regmean.py","file_url":"https://github.com/huangcb01/HARC/blob/HEAD/src/merge_method/regmean.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0d2c2dfbee2f518c","mcp_get_code":{"code_sha256":"0d2c2dfbee2f518c"}},{"arxiv_id":"2605.30794","paper":"/paper/arxiv-2605-30794","title":"MechVQA: Benchmarking and Enhancing Multimodal LLMs on Comprehensive Mechanical Drawing Understanding","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"xiaofengShi/MechVQA","path":"evaluation/mechvqa_eval/io_utils.py","file_url":"https://github.com/xiaofengShi/MechVQA/blob/HEAD/evaluation/mechvqa_eval/io_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0596526ef5c6ca8e","mcp_get_code":{"code_sha256":"0596526ef5c6ca8e"}},{"arxiv_id":"2605.28837","paper":"/paper/arxiv-2605-28837","title":"SERC: LDPC-Inspired Semantic Error Correction for Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"labhai/SERC","path":"src/utils.py","file_url":"https://github.com/labhai/SERC/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2d3cbbede5014185","mcp_get_code":{"code_sha256":"2d3cbbede5014185"}},{"arxiv_id":"2605.26315","paper":"/paper/arxiv-2605-26315","title":"Curriculum Learning for Safety Alignment","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Sandeep5500/curriculum-learning-for-safety","path":"src/phase1/create_curriculum.py","file_url":"https://github.com/Sandeep5500/curriculum-learning-for-safety/blob/HEAD/src/phase1/create_curriculum.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"12db38ad37394041","mcp_get_code":{"code_sha256":"12db38ad37394041"}},{"arxiv_id":"2605.25920","paper":"/paper/arxiv-2605-25920","title":"Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"AlexFanw/LegalSearch-R1","path":"user/legalsearch_data_process.py","file_url":"https://github.com/AlexFanw/LegalSearch-R1/blob/HEAD/user/legalsearch_data_process.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0eed8cf50397ec71","mcp_get_code":{"code_sha256":"0eed8cf50397ec71"}},{"arxiv_id":"2605.23918","paper":"/paper/arxiv-2605-23918","title":"The Model Parking Tax: Quantifying the Hidden Energy Cost of Always-On GPU Model Deployment","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"8bitai/gpu-parking-tax","path":"analysis/generate_new_experiments_figures.py","file_url":"https://github.com/8bitai/gpu-parking-tax/blob/HEAD/analysis/generate_new_experiments_figures.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"08acd783abc2687d","mcp_get_code":{"code_sha256":"08acd783abc2687d"}},{"arxiv_id":"2605.22064","paper":"/paper/arxiv-2605-22064","title":"Hy-MT2: A Family of Fast, Efficient and Powerful Multilingual Translation Models in the Wild","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Tencent-Hunyuan/Hy-MT2","path":"IFMTBench/run_eval.py","file_url":"https://github.com/Tencent-Hunyuan/Hy-MT2/blob/HEAD/IFMTBench/run_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"3b68b7b1239b923f","mcp_get_code":{"code_sha256":"3b68b7b1239b923f"}},{"arxiv_id":"2605.11134","paper":"/paper/arxiv-2605-11134","title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"cmoyacal/tie-training","path":"src/llm/concat_for_training.py","file_url":"https://github.com/cmoyacal/tie-training/blob/HEAD/src/llm/concat_for_training.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4ea2119c8445dd18","mcp_get_code":{"code_sha256":"4ea2119c8445dd18"}},{"arxiv_id":"2605.10186","paper":"/paper/arxiv-2605-10186","title":"LegalCiteBench: Evaluating Citation Reliability in Legal Language Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Sijia711/LegalCiteBench","path":"legal-citation-benchmark-clean/analysis/run_prompt_mitigation.py","file_url":"https://github.com/Sijia711/LegalCiteBench/blob/HEAD/legal-citation-benchmark-clean/analysis/run_prompt_mitigation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c06818718c5e5355","mcp_get_code":{"code_sha256":"c06818718c5e5355"}},{"arxiv_id":"2605.08333","paper":"/paper/arxiv-2605-08333","title":"CDS4RAG: Cyclic Dual-Sequential Hyperparameter Optimization for RAG","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"ideas-labo/cds4rag","path":"Run_util.py","file_url":"https://github.com/ideas-labo/cds4rag/blob/HEAD/Run_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4561fb8f5d531d34","mcp_get_code":{"code_sha256":"4561fb8f5d531d34"}},{"arxiv_id":"2605.03824","paper":"/paper/arxiv-2605-03824","title":"Reproducing Complex Set-Compositional Information Retrieval","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"informagi/Complex-Set-Compositional-IR","path":"code/cost_estimates/estimate_reranking_costs.py","file_url":"https://github.com/informagi/Complex-Set-Compositional-IR/blob/HEAD/code/cost_estimates/estimate_reranking_costs.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a5175d986e184179","mcp_get_code":{"code_sha256":"a5175d986e184179"}},{"arxiv_id":"2604.27249","paper":"/paper/arxiv-2604-27249","title":"Instruction Complexity Induces Positional Collapse in Adversarial LLM Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"synthiumjp/bcb-sandbagging-pilot","path":"generate_figures.py","file_url":"https://github.com/synthiumjp/bcb-sandbagging-pilot/blob/HEAD/generate_figures.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2467069a4724a8c1","mcp_get_code":{"code_sha256":"2467069a4724a8c1"}},{"arxiv_id":"2604.25847","paper":"/paper/arxiv-2604-25847","title":"From Soliloquy to Agora: Memory-Enhanced LLM Agents with Decentralized Debate for Optimization Modeling","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"CHIANGEL/Agora-Opt","path":"code/Agora-Opt/src/debate_memory/augment_memory_from_standalone_runs.py","file_url":"https://github.com/CHIANGEL/Agora-Opt/blob/HEAD/code/Agora-Opt/src/debate_memory/augment_memory_from_standalone_runs.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"888f5c53e1db5aa6","mcp_get_code":{"code_sha256":"888f5c53e1db5aa6"}},{"arxiv_id":"2604.25847","paper":"/paper/arxiv-2604-25847","title":"From Soliloquy to Agora: Memory-Enhanced LLM Agents with Decentralized Debate for Optimization Modeling","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"CHIANGEL/Agora-Opt","path":"code/Agora-Opt/src/debate_memory/debate_memory_builder.py","file_url":"https://github.com/CHIANGEL/Agora-Opt/blob/HEAD/code/Agora-Opt/src/debate_memory/debate_memory_builder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"70016d2a0a8b58c9","mcp_get_code":{"code_sha256":"70016d2a0a8b58c9"}},{"arxiv_id":"2604.19185","paper":"/paper/arxiv-2604-19185","title":"SCURank: Ranking Multiple Candidate Summaries with Summary Content Units for Enhanced Summarization","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"IKMLab/SCURank","path":"utils.py","file_url":"https://github.com/IKMLab/SCURank/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d18bc6c6c02cdbb2","mcp_get_code":{"code_sha256":"d18bc6c6c02cdbb2"}},{"arxiv_id":"2604.19185","paper":"/paper/arxiv-2604-19185","title":"SCURank: Ranking Multiple Candidate Summaries with Summary Content Units for Enhanced Summarization","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"IKMLab/SCURank","path":"experiments/human-compare/data_build.py","file_url":"https://github.com/IKMLab/SCURank/blob/HEAD/experiments/human-compare/data_build.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bda0f2927436562b","mcp_get_code":{"code_sha256":"bda0f2927436562b"}},{"arxiv_id":"2604.16763","paper":"/paper/arxiv-2604-16763","title":"LLM-Extracted Covariates for Clinical Causal Inference: Rethinking Integration Strategies","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"fpxlei/LLM-Covariates-Causal","path":"finetune/train_lora.py","file_url":"https://github.com/fpxlei/LLM-Covariates-Causal/blob/HEAD/finetune/train_lora.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1bfdd49cdba2faac","mcp_get_code":{"code_sha256":"1bfdd49cdba2faac"}},{"arxiv_id":"2604.16349","paper":"/paper/arxiv-2604-16349","title":"Benchmarking Real-Time Question Answering via Executable Code Workflows","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"leaves-slient/RT-Bench","path":"DeepResearch/evaluation/evaluate_hle_official.py","file_url":"https://github.com/leaves-slient/RT-Bench/blob/HEAD/DeepResearch/evaluation/evaluate_hle_official.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e587e0fad56e82f2","mcp_get_code":{"code_sha256":"e587e0fad56e82f2"}},{"arxiv_id":"2604.11048","paper":"/paper/arxiv-2604-11048","title":"A Systematic Analysis of the Impact of Persona Steering on LLM Capabilities","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"cjia7/DPR","path":"src/npti/eval/eval_bbh.py","file_url":"https://github.com/cjia7/DPR/blob/HEAD/src/npti/eval/eval_bbh.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8fe69f136a96c214","mcp_get_code":{"code_sha256":"8fe69f136a96c214"}},{"arxiv_id":"2604.11048","paper":"/paper/arxiv-2604-11048","title":"A Systematic Analysis of the Impact of Persona Steering on LLM Capabilities","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"cjia7/DPR","path":"src/npti/eval/gpt4_score.py","file_url":"https://github.com/cjia7/DPR/blob/HEAD/src/npti/eval/gpt4_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c7a248507261f132","mcp_get_code":{"code_sha256":"c7a248507261f132"}},{"arxiv_id":"2604.07119","paper":"/paper/arxiv-2604-07119","title":"Are Non-English Papers Reviewed Fairly? Language-of-Study Bias in NLP Peer Reviews","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"GGLAB-KU/LOBSTER","path":"base_runner.py","file_url":"https://github.com/GGLAB-KU/LOBSTER/blob/HEAD/base_runner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"478febea332e6ec2","mcp_get_code":{"code_sha256":"478febea332e6ec2"}},{"arxiv_id":"2604.05114","paper":"/paper/arxiv-2604-05114","title":"π 2 : Structure-Originated Reasoning Data Improves Long-Context Reasoning Ability of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"vt-pi-squared/pi-squared","path":"datasets/ours/code/v1_merge_and_postprocess_full_data_pipeline.py","file_url":"https://github.com/vt-pi-squared/pi-squared/blob/HEAD/datasets/ours/code/v1_merge_and_postprocess_full_data_pipeline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bbf4c6a9a91e3bea","mcp_get_code":{"code_sha256":"bbf4c6a9a91e3bea"}},{"arxiv_id":"2604.04017","paper":"/paper/arxiv-2604-04017","title":"GeoBrowse: A Geolocation Benchmark for Agentic Tool Use with Expert-Annotated Reasoning Traces","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ornamentt/GeoBrowse","path":"Evaluation/baseline/evaluate.py","file_url":"https://github.com/ornamentt/GeoBrowse/blob/HEAD/Evaluation/baseline/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e587e0fad56e82f2","mcp_get_code":{"code_sha256":"e587e0fad56e82f2"}},{"arxiv_id":"2604.00536","paper":"/paper/arxiv-2604-00536","title":"Optimsyn: Influence-Guided Rubrics Optimization for Synthetic Data Generation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"FanZT6/OptimSyn","path":"data_synthesis/generate_qa.py","file_url":"https://github.com/FanZT6/OptimSyn/blob/HEAD/data_synthesis/generate_qa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fc7b419016b1e837","mcp_get_code":{"code_sha256":"fc7b419016b1e837"}},{"arxiv_id":"2604.00536","paper":"/paper/arxiv-2604-00536","title":"Optimsyn: Influence-Guided Rubrics Optimization for Synthetic Data Generation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"FanZT6/OptimSyn","path":"data_synthesis/generate_qa_eval.py","file_url":"https://github.com/FanZT6/OptimSyn/blob/HEAD/data_synthesis/generate_qa_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"761da305e0f257c4","mcp_get_code":{"code_sha256":"761da305e0f257c4"}},{"arxiv_id":"2604.00536","paper":"/paper/arxiv-2604-00536","title":"Optimsyn: Influence-Guided Rubrics Optimization for Synthetic Data Generation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"FanZT6/OptimSyn","path":"data_synthesis/generate_rubrics.py","file_url":"https://github.com/FanZT6/OptimSyn/blob/HEAD/data_synthesis/generate_rubrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"131f8568b132a083","mcp_get_code":{"code_sha256":"131f8568b132a083"}},{"arxiv_id":"2603.23136","paper":"/paper/arxiv-2603-23136","title":"HGNet: Scalable Foundation Model for Automated Knowledge Graph Generation from Scientific Literature","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"basiralab/HGNet","path":"datasets/SPHERE/prepare_sphere.py","file_url":"https://github.com/basiralab/HGNet/blob/HEAD/datasets/SPHERE/prepare_sphere.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d796bc0f8d962649","mcp_get_code":{"code_sha256":"d796bc0f8d962649"}},{"arxiv_id":"2603.10992","paper":"/paper/arxiv-2603-10992","title":"A Tutorial Review of Bayesian Optimization with Gaussian Processes to Accelerate Stationary Point Searches","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"lode-org/ChemGP","path":"benchmarks/analysis/summarize.py","file_url":"https://github.com/lode-org/ChemGP/blob/HEAD/benchmarks/analysis/summarize.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6924be70a5282217","mcp_get_code":{"code_sha256":"6924be70a5282217"}},{"arxiv_id":"2603.03884","paper":"/paper/arxiv-2603-03884","title":"CzechTopic: A Benchmark for Zero-Shot Topic Localization in Historical Czech Documents","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"dcgm/czechtopic","path":"evaluation/common.py","file_url":"https://github.com/dcgm/czechtopic/blob/HEAD/evaluation/common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0a1da25975c47643","mcp_get_code":{"code_sha256":"0a1da25975c47643"}},{"arxiv_id":"2602.14594","paper":"/paper/arxiv-2602-14594","title":"The Wikidata Query Logs Dataset","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ad-freiburg/wikidata-query-logs","path":"check_overlap.py","file_url":"https://github.com/ad-freiburg/wikidata-query-logs/blob/HEAD/check_overlap.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3ad6b4253f5b5e42","mcp_get_code":{"code_sha256":"3ad6b4253f5b5e42"}},{"arxiv_id":"2602.13576","paper":"/paper/arxiv-2602-13576","title":"Rubrics as an Attack Surface: Stealthy Preference Drift in LLM Judges","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ZDCSlab/Rubrics-as-an-Attack-Surface","path":"downstream_eval/eval/utils_pairwise.py","file_url":"https://github.com/ZDCSlab/Rubrics-as-an-Attack-Surface/blob/HEAD/downstream_eval/eval/utils_pairwise.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"400eeff90e1ba381","mcp_get_code":{"code_sha256":"400eeff90e1ba381"}},{"arxiv_id":"2602.07422","paper":"/paper/arxiv-2602-07422","title":"Secure Code Generation via Online Reinforcement Learning with Vulnerability Reward Model","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"AndrewWTY/SecCoderX","path":"vul_induce_prompt_pipeline/inference_instructions_with_vllm.py","file_url":"https://github.com/AndrewWTY/SecCoderX/blob/HEAD/vul_induce_prompt_pipeline/inference_instructions_with_vllm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0348ea9288baf003","mcp_get_code":{"code_sha256":"0348ea9288baf003"}},{"arxiv_id":"2602.05910","paper":"/paper/arxiv-2602-05910","title":"Chunky Post-Training: Data Driven Failures of Generalization","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"seoirsem/SURF","path":"surf/core/streaming.py","file_url":"https://github.com/seoirsem/SURF/blob/HEAD/surf/core/streaming.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b4599e397fec376","mcp_get_code":{"code_sha256":"0b4599e397fec376"}},{"arxiv_id":"2601.22162","paper":"/paper/arxiv-2601-22162","title":"UniFinEval: Towards Unified Evaluation of Financial Multimodal Models across Text, Images and Videos","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"aifinlab/UniFinEval","path":"evaluate_py/data_loader.py","file_url":"https://github.com/aifinlab/UniFinEval/blob/HEAD/evaluate_py/data_loader.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fca995540b6041ae","mcp_get_code":{"code_sha256":"fca995540b6041ae"}},{"arxiv_id":"2601.22047","paper":"/paper/arxiv-2601-22047","title":"On the Paradoxical Interference between Instruction-Following and Task Solving","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"kijlk/IF-Interference","path":"src/math_and_qa/math_utils.py","file_url":"https://github.com/kijlk/IF-Interference/blob/HEAD/src/math_and_qa/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"505e67e438e9eb19","mcp_get_code":{"code_sha256":"505e67e438e9eb19"}},{"arxiv_id":"2601.17197","paper":"/paper/arxiv-2601-17197","title":"Reasoning Beyond Literal: Cross-style Multimodal Reasoning for Figurative Language Understanding","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"scheshmi/CrossStyle-MMR","path":"train/sft_combined.py","file_url":"https://github.com/scheshmi/CrossStyle-MMR/blob/HEAD/train/sft_combined.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c9bf1dffc4d2016c","mcp_get_code":{"code_sha256":"c9bf1dffc4d2016c"}},{"arxiv_id":"2601.16449","paper":"/paper/arxiv-2601-16449","title":"Emotion-LLaMAv2 and MMEVerse: A New Framework and Benchmark for Multimodal Emotion Understanding","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ooochen-30/Emotion-LLaMA-v2","path":"evaluation/score_split.py","file_url":"https://github.com/ooochen-30/Emotion-LLaMA-v2/blob/HEAD/evaluation/score_split.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"a118c735eadfbee8","mcp_get_code":{"code_sha256":"a118c735eadfbee8"}},{"arxiv_id":"2601.16449","paper":"/paper/arxiv-2601-16449","title":"Emotion-LLaMAv2 and MMEVerse: A New Framework and Benchmark for Multimodal Emotion Understanding","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ooochen-30/Emotion-LLaMA-v2","path":"evaluation/score_split_sentiment_no_neutral.py","file_url":"https://github.com/ooochen-30/Emotion-LLaMA-v2/blob/HEAD/evaluation/score_split_sentiment_no_neutral.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"af8fdb6ca114a86c","mcp_get_code":{"code_sha256":"af8fdb6ca114a86c"}},{"arxiv_id":"2512.14681","paper":"/paper/arxiv-2512-14681","title":"Fast and Accurate Causal Parallel Decoding using Jacobi Forcing FAST AND ACCURATE CAUSAL PARALLEL DECODING USING JACOBI FORCING","date":null,"month_inferred_from_arxiv_id":"2025-12","title_source":"syntology","repo":"hao-ai-lab/JacobiForcing","path":"JacobiForcing/ar_inference_baseline.py","file_url":"https://github.com/hao-ai-lab/JacobiForcing/blob/HEAD/JacobiForcing/ar_inference_baseline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"163eb1381b9a92fb","mcp_get_code":{"code_sha256":"163eb1381b9a92fb"}},{"arxiv_id":"2510.07364","paper":"/paper/arxiv-2510-07364","title":"Base Models Know How to Reason, Thinking Models Learn When","date":"2025-10-08","month_inferred_from_arxiv_id":null,"title_source":"syntology","repo":"cvenhoff/thinking-llms-interp","path":"human_eval/sample.py","file_url":"https://github.com/cvenhoff/thinking-llms-interp/blob/HEAD/human_eval/sample.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"80fe4b7f121fc9b9","mcp_get_code":{"code_sha256":"80fe4b7f121fc9b9"}},{"arxiv_id":"2509.16598","paper":"/paper/arxiv-2509-16598","title":"PruneCD: Contrasting Pruned Self Model to Improve Decoding Factuality","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"hoeng4/PruneCD","path":"2_benchmark/5_strqa.py","file_url":"https://github.com/hoeng4/PruneCD/blob/HEAD/2_benchmark/5_strqa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ff3b5ca99f326851","mcp_get_code":{"code_sha256":"ff3b5ca99f326851"}},{"arxiv_id":"2509.16598","paper":"/paper/arxiv-2509-16598","title":"PruneCD: Contrasting Pruned Self Model to Improve Decoding Factuality","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"hoeng4/PruneCD","path":"2_benchmark/6_gsm8k.py","file_url":"https://github.com/hoeng4/PruneCD/blob/HEAD/2_benchmark/6_gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b5d313d6f48bb56a","mcp_get_code":{"code_sha256":"b5d313d6f48bb56a"}},{"arxiv_id":"2509.16112","paper":"/paper/arxiv-2509-16112","title":"CodeRAG: Finding Relevant and Necessary Knowledge for Retrieval-Augmented Repository-Level Code Completion","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"KDEGroup/CodeRAG","path":"coderag/benchmark/common.py","file_url":"https://github.com/KDEGroup/CodeRAG/blob/HEAD/coderag/benchmark/common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"08e488a1ef10ab01","mcp_get_code":{"code_sha256":"08e488a1ef10ab01"}},{"arxiv_id":"2509.13347","paper":"/paper/arxiv-2509-13347","title":"OpenHA: A Series of Open-Source Hierarchical Agentic Models in Minecraft","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"CraftJarvis/OpenHA","path":"openagents/utils/file_op.py","file_url":"https://github.com/CraftJarvis/OpenHA/blob/HEAD/openagents/utils/file_op.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3ac774349fa56ea5","mcp_get_code":{"code_sha256":"3ac774349fa56ea5"}},{"arxiv_id":"2509.01907","paper":"/paper/arxiv-2509-01907","title":"RSCC: A Large-Scale Remote Sensing Change Caption Dataset for Disaster Events","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"Bili-Sakura/RSCC","path":"evaluation/metrics.py","file_url":"https://github.com/Bili-Sakura/RSCC/blob/HEAD/evaluation/metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e7b871016c230aac","mcp_get_code":{"code_sha256":"e7b871016c230aac"}},{"arxiv_id":"2507.02592","paper":"/paper/websailor-navigating-super-human-reasoning","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","date":"2025-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba-nlp/webwalker","path":"evaluation/evaluate_hle_official.py","file_url":"https://github.com/alibaba-nlp/webwalker/blob/HEAD/evaluation/evaluate_hle_official.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e587e0fad56e82f2","mcp_get_code":{"code_sha256":"e587e0fad56e82f2"}},{"arxiv_id":"2506.02973","paper":"/paper/expanding-before-inferring-enhancing","title":"Expanding before Inferring: Enhancing Factuality in Large Language Models through Premature Layers Interpolation","date":"2025-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CuSO4-Chen/PLI","path":"src/evaluation/gsm8k_eval.py","file_url":"https://github.com/CuSO4-Chen/PLI/blob/HEAD/src/evaluation/gsm8k_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e9a7befa1d14a877","mcp_get_code":{"code_sha256":"e9a7befa1d14a877"}},{"arxiv_id":"2506.00391","paper":"/paper/share-an-slm-based-hierarchical-action","title":"SHARE: An SLM-based Hierarchical Action CorREction Assistant for Text-to-SQL","date":"2025-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"quge2023/SHARE","path":"src/utils.py","file_url":"https://github.com/quge2023/SHARE/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32da31b37eea7836","mcp_get_code":{"code_sha256":"32da31b37eea7836"}},{"arxiv_id":"2505.24787","paper":"/paper/draw-all-your-imagine-a-holistic-benchmark","title":"Draw ALL Your Imagine: A Holistic Benchmark and Agent Framework for Complex Instruction-based Image Generation","date":"2025-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yczhou001/longbench-t2i","path":"utils/utils.py","file_url":"https://github.com/yczhou001/longbench-t2i/blob/HEAD/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fdea7da6487b6a46","mcp_get_code":{"code_sha256":"fdea7da6487b6a46"}},{"arxiv_id":"2505.20302","paper":"/paper/verithoughts-enabling-automated-verilog-code","title":"VeriThoughts: Enabling Automated Verilog Code Generation using Reasoning and Formal Verification","date":"2025-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wilyub/verithoughts","path":"verithoughts/multi-turn-yosys-generation.py","file_url":"https://github.com/wilyub/verithoughts/blob/HEAD/verithoughts/multi-turn-yosys-generation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4398a8aef9b9aba0","mcp_get_code":{"code_sha256":"4398a8aef9b9aba0"}},{"arxiv_id":"2505.12531","paper":"/paper/esc-judge-a-framework-for-comparing-emotional","title":"ESC-Judge: A Framework for Comparing Emotional Support Conversational Agents","date":"2025-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"navidmdn/ESC-Judge","path":"multidim_judge_from_merged.py","file_url":"https://github.com/navidmdn/ESC-Judge/blob/HEAD/multidim_judge_from_merged.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"720f3f0012bdfd66","mcp_get_code":{"code_sha256":"720f3f0012bdfd66"}},{"arxiv_id":"2505.12531","paper":"/paper/esc-judge-a-framework-for-comparing-emotional","title":"ESC-Judge: A Framework for Comparing Emotional Support Conversational Agents","date":"2025-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"navidmdn/ESC-Judge","path":"multidim_geval.py","file_url":"https://github.com/navidmdn/ESC-Judge/blob/HEAD/multidim_geval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"885cdf0c8392f181","mcp_get_code":{"code_sha256":"885cdf0c8392f181"}},{"arxiv_id":"2505.10557","paper":"/paper/mathcoder-vl-bridging-vision-and-code-for","title":"MathCoder-VL: Bridging Vision and Code for Enhanced Multimodal Mathematical Reasoning","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"17081a7b73a41850","mcp_get_code":{"code_sha256":"17081a7b73a41850"}},{"arxiv_id":"2505.02881","paper":"/paper/rewriting-pre-training-data-boosts-llm","title":"Rewriting Pre-Training Data Boosts LLM Performance in Math and Code","date":"2025-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rioyokotalab/swallow-code-math","path":"src/math/finemath-4+-rewrite-v1.py","file_url":"https://github.com/rioyokotalab/swallow-code-math/blob/HEAD/src/math/finemath-4%2B-rewrite-v1.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0d9c05c05e8d3560","mcp_get_code":{"code_sha256":"0d9c05c05e8d3560"}},{"arxiv_id":"2504.21463","paper":"/paper/rwkv-x-a-linear-complexity-hybrid-language","title":"RWKV-X: A Linear Complexity Hybrid Language Model","date":"2025-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"howard-hou/rwkv-x","path":"evaluation/eval_long_loss.py","file_url":"https://github.com/howard-hou/rwkv-x/blob/HEAD/evaluation/eval_long_loss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"933d4bd3c4bd2d14","mcp_get_code":{"code_sha256":"933d4bd3c4bd2d14"}},{"arxiv_id":"2504.07521","paper":"/paper/why-we-feel-breaking-boundaries-in-emotional","title":"Why We Feel: Breaking Boundaries in Emotional Reasoning with Multimodal Large Language Models","date":"2025-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lum1104/eibench","path":"EIBench/human_eval/web_ann_basic.py","file_url":"https://github.com/lum1104/eibench/blob/HEAD/EIBench/human_eval/web_ann_basic.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e85a5515b1b5db8b","mcp_get_code":{"code_sha256":"e85a5515b1b5db8b"}},{"arxiv_id":"2504.04635","paper":null,"title":"arXiv:2504.04635","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"patqdasilva/steering-off-course","path":"DoLa/strqa_eval.py","file_url":"https://github.com/patqdasilva/steering-off-course/blob/HEAD/DoLa/strqa_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ff3b5ca99f326851","mcp_get_code":{"code_sha256":"ff3b5ca99f326851"}},{"arxiv_id":"2504.04635","paper":null,"title":"arXiv:2504.04635","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"patqdasilva/steering-off-course","path":"DoLa/gsm8k_eval.py","file_url":"https://github.com/patqdasilva/steering-off-course/blob/HEAD/DoLa/gsm8k_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b5d313d6f48bb56a","mcp_get_code":{"code_sha256":"b5d313d6f48bb56a"}},{"arxiv_id":"2503.16929","paper":"/paper/temple-temporal-preference-learning-of-video","title":"TEMPLE:Temporal Preference Learning of Video LLMs via Difficulty Scheduling and Pre-SFT Alignment","date":"2025-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lscpku/temple","path":"preprocess.py","file_url":"https://github.com/lscpku/temple/blob/HEAD/preprocess.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"db41886fef3918ad","mcp_get_code":{"code_sha256":"db41886fef3918ad"}},{"arxiv_id":"2503.16212","paper":"/paper/mathfusion-enhancing-mathematic-problem","title":"MathFusion: Enhancing Mathematic Problem-solving of LLM through Instruction Fusion","date":"2025-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qizhipei/mathfusion","path":"evaluation/dart_math/utils.py","file_url":"https://github.com/qizhipei/mathfusion/blob/HEAD/evaluation/dart_math/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fb57a712227893bd","mcp_get_code":{"code_sha256":"fb57a712227893bd"}},{"arxiv_id":"2503.14443","paper":"/paper/envbench-a-benchmark-for-automated","title":"EnvBench: A Benchmark for Automated Environment Setup","date":"2025-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JetBrains-Research/EnvBench","path":"env_setup_utils/analysis/analysis_utils.py","file_url":"https://github.com/JetBrains-Research/EnvBench/blob/HEAD/env_setup_utils/analysis/analysis_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7035f0d171657373","mcp_get_code":{"code_sha256":"7035f0d171657373"}},{"arxiv_id":"2503.14443","paper":"/paper/envbench-a-benchmark-for-automated","title":"EnvBench: A Benchmark for Automated Environment Setup","date":"2025-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JetBrains-Research/EnvBench","path":"env_setup_utils/analysis/analyze_results.py","file_url":"https://github.com/JetBrains-Research/EnvBench/blob/HEAD/env_setup_utils/analysis/analyze_results.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"97ada5f0eb559039","mcp_get_code":{"code_sha256":"97ada5f0eb559039"}},{"arxiv_id":"2503.14443","paper":"/paper/envbench-a-benchmark-for-automated","title":"EnvBench: A Benchmark for Automated Environment Setup","date":"2025-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JetBrains-Research/EnvBench","path":"env_setup_utils/analysis/scripts_viewer.py","file_url":"https://github.com/JetBrains-Research/EnvBench/blob/HEAD/env_setup_utils/analysis/scripts_viewer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"db2166cef66a916e","mcp_get_code":{"code_sha256":"db2166cef66a916e"}},{"arxiv_id":"2503.10582","paper":"/paper/visualwebinstruct-scaling-up-multimodal","title":"VisualWebInstruct: Scaling up Multimodal Instruction Data through Web Search","date":"2025-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tiger-ai-lab/visualwebinstruct","path":"VisualWebInstruct/answer_alignment.py","file_url":"https://github.com/tiger-ai-lab/visualwebinstruct/blob/HEAD/VisualWebInstruct/answer_alignment.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd1b2e17d5682457","mcp_get_code":{"code_sha256":"cd1b2e17d5682457"}},{"arxiv_id":"2503.07539","paper":"/paper/xifbench-evaluating-large-language-models-on","title":"XIFBench: Evaluating Large Language Models on Multilingual Instruction Following","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhenyuli801/XIFBench","path":"A1_sample_instructions.py","file_url":"https://github.com/zhenyuli801/XIFBench/blob/HEAD/A1_sample_instructions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d4c04971c3f6dc69","mcp_get_code":{"code_sha256":"d4c04971c3f6dc69"}},{"arxiv_id":"2503.07459","paper":"/paper/medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gersteinlab/medagents-benchmark","path":"output/utils.py","file_url":"https://github.com/gersteinlab/medagents-benchmark/blob/HEAD/output/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b07109264c29d074","mcp_get_code":{"code_sha256":"b07109264c29d074"}},{"arxiv_id":"2503.07265","paper":"/paper/wise-a-world-knowledge-informed-semantic","title":"WISE: A World Knowledge-Informed Semantic Evaluation for Text-to-Image Generation","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PKU-YuanGroup/WISE","path":"vllm_eval.py","file_url":"https://github.com/PKU-YuanGroup/WISE/blob/HEAD/vllm_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0bc2febf8ef317cf","mcp_get_code":{"code_sha256":"0bc2febf8ef317cf"}},{"arxiv_id":"2503.06800","paper":"/paper/videophy-2-a-challenging-action-centric","title":"VideoPhy-2: A Challenging Action-Centric Physical Commonsense Evaluation in Video Generation","date":"2025-03-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Hritikbansal/videophy","path":"VIDEOPHY2/data_utils/xgpt3_dataset.py","file_url":"https://github.com/Hritikbansal/videophy/blob/HEAD/VIDEOPHY2/data_utils/xgpt3_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3a23711c5501382","mcp_get_code":{"code_sha256":"f3a23711c5501382"}},{"arxiv_id":"2503.04800","paper":"/paper/hoh-a-dynamic-benchmark-for-evaluating-the","title":"HoH: A Dynamic Benchmark for Evaluating the Impact of Outdated Information on Retrieval-Augmented Generation","date":"2025-03-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"0russwest0/HoH","path":"qa_generate/src/utils.py","file_url":"https://github.com/0russwest0/HoH/blob/HEAD/qa_generate/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"372d16df79db74a4","mcp_get_code":{"code_sha256":"372d16df79db74a4"}},{"arxiv_id":"2502.17424","paper":"/paper/emergent-misalignment-narrow-finetuning-can","title":"Emergent Misalignment: Narrow finetuning can produce broadly misaligned LLMs","date":"2025-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"emergent-misalignment/emergent-misalignment","path":"open_models/utils.py","file_url":"https://github.com/emergent-misalignment/emergent-misalignment/blob/HEAD/open_models/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1d683a1b53d1615a","mcp_get_code":{"code_sha256":"1d683a1b53d1615a"}},{"arxiv_id":"2502.14768","paper":"/paper/logic-rl-unleashing-llm-reasoning-with-rule","title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Unakar/Logic-RL","path":"eval_kk/main_eval_instruct.py","file_url":"https://github.com/Unakar/Logic-RL/blob/HEAD/eval_kk/main_eval_instruct.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0998b08e39672d66","mcp_get_code":{"code_sha256":"0998b08e39672d66"}},{"arxiv_id":"2502.12067","paper":"/paper/tokenskip-controllable-chain-of-thought","title":"TokenSkip: Controllable Chain-of-Thought Compression in LLMs","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hemingkx/TokenSkip","path":"LLMLingua.py","file_url":"https://github.com/hemingkx/TokenSkip/blob/HEAD/LLMLingua.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3dc03e3fd9c53b25","mcp_get_code":{"code_sha256":"3dc03e3fd9c53b25"}},{"arxiv_id":"2502.06205","paper":"/paper/c-3po-compact-plug-and-play-proxy","title":"C-3PO: Compact Plug-and-Play Proxy Optimization to Achieve Human-like Retrieval-Augmented Generation","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Chen-GX/C-3PO","path":"C-3PO/utils.py","file_url":"https://github.com/Chen-GX/C-3PO/blob/HEAD/C-3PO/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b27698eecee6ba64","mcp_get_code":{"code_sha256":"b27698eecee6ba64"}},{"arxiv_id":"2502.01976","paper":"/paper/citer-collaborative-inference-for-efficient","title":"CITER: Collaborative Inference for Efficient Large Language Model Decoding with Token-Level Routing","date":"2025-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aiming-lab/CITER","path":"src/pipeline/token_route.py","file_url":"https://github.com/aiming-lab/CITER/blob/HEAD/src/pipeline/token_route.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0bdfccf9d2a4cbc6","mcp_get_code":{"code_sha256":"0bdfccf9d2a4cbc6"}},{"arxiv_id":"2501.12599","paper":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/math-v","path":"models/utils.py","file_url":"https://github.com/mathllm/math-v/blob/HEAD/models/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"17081a7b73a41850","mcp_get_code":{"code_sha256":"17081a7b73a41850"}},{"arxiv_id":"2501.12599","paper":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/math-v","path":"models/GPT_with_caption.py","file_url":"https://github.com/mathllm/math-v/blob/HEAD/models/GPT_with_caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8bcc3df886c26873","mcp_get_code":{"code_sha256":"8bcc3df886c26873"}},{"arxiv_id":"2501.12599","paper":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/math-v","path":"models/GPT4.py","file_url":"https://github.com/mathllm/math-v/blob/HEAD/models/GPT4.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f57347eff09ff0f5","mcp_get_code":{"code_sha256":"f57347eff09ff0f5"}},{"arxiv_id":"2412.09413","paper":"/paper/imitate-explore-and-self-improve-a","title":"Imitate, Explore, and Self-Improve: A Reproduction Report on Slow-thinking Reasoning Systems","date":"2024-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RUCAIBox/Slow_Thinking_with_LLMs","path":"STILL-3-TOOL/data_synthesis/coding_by_thinking.py","file_url":"https://github.com/RUCAIBox/Slow_Thinking_with_LLMs/blob/HEAD/STILL-3-TOOL/data_synthesis/coding_by_thinking.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4b0aef908978da14","mcp_get_code":{"code_sha256":"4b0aef908978da14"}},{"arxiv_id":"2412.06660","paper":"/paper/mumu-llama-multi-modal-music-understanding","title":"MuMu-LLaMA: Multi-modal Music Understanding and Generation via Large Language Models","date":null,"month_inferred_from_arxiv_id":"2024-12","title_source":"archive","repo":"shansongliu/M2UGen","path":"DataSet/MUImage/llava_caption.py","file_url":"https://github.com/shansongliu/M2UGen/blob/HEAD/DataSet/MUImage/llava_caption.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9eb0b14fc790de41","mcp_get_code":{"code_sha256":"9eb0b14fc790de41"}},{"arxiv_id":"2412.06141","paper":"/paper/mmedpo-aligning-medical-vision-language","title":"MMedPO: Aligning Medical Vision-Language Models with Clinical-Aware Multimodal Preference Optimization","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aiming-lab/mmedpo","path":"eval/eval_report.py","file_url":"https://github.com/aiming-lab/mmedpo/blob/HEAD/eval/eval_report.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77678f2758141df9","mcp_get_code":{"code_sha256":"77678f2758141df9"}},{"arxiv_id":"2412.02158","paper":"/paper/agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kki2eve/agri-llava","path":"agri_llava/eval/run_eval.py","file_url":"https://github.com/kki2eve/agri-llava/blob/HEAD/agri_llava/eval/run_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77678f2758141df9","mcp_get_code":{"code_sha256":"77678f2758141df9"}},{"arxiv_id":"2411.02433","paper":"/paper/sled-self-logits-evolution-decoding-for","title":"SLED: Self Logits Evolution Decoding for Improving Factuality in Large Language Models","date":"2024-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JayZhang42/SLED","path":"utils/utils_gsm8k.py","file_url":"https://github.com/JayZhang42/SLED/blob/HEAD/utils/utils_gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"999e06e03a65eb6d","mcp_get_code":{"code_sha256":"999e06e03a65eb6d"}},{"arxiv_id":"2411.02433","paper":"/paper/sled-self-logits-evolution-decoding-for","title":"SLED: Self Logits Evolution Decoding for Improving Factuality in Large Language Models","date":"2024-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JayZhang42/SLED","path":"utils/utils_strqa.py","file_url":"https://github.com/JayZhang42/SLED/blob/HEAD/utils/utils_strqa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"21ebee30ce3d7085","mcp_get_code":{"code_sha256":"21ebee30ce3d7085"}},{"arxiv_id":"2411.01855","paper":"/paper/can-language-models-learn-to-skip-steps","title":"Can Language Models Learn to Skip Steps?","date":"2024-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tengxiaoliu/LM_skip","path":"src/evaluate_aoa.py","file_url":"https://github.com/tengxiaoliu/LM_skip/blob/HEAD/src/evaluate_aoa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"81f1e40889f24580","mcp_get_code":{"code_sha256":"81f1e40889f24580"}},{"arxiv_id":"2410.12705","paper":"/paper/worldcuisines-a-massive-scale-benchmark-for","title":"WorldCuisines: A Massive-Scale Benchmark for Multilingual and Multicultural Visual Question Answering on Global Cuisines","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"worldcuisines/worldcuisines","path":"evaluation/score/score.py","file_url":"https://github.com/worldcuisines/worldcuisines/blob/HEAD/evaluation/score/score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bf44cfc1771f959a","mcp_get_code":{"code_sha256":"bf44cfc1771f959a"}},{"arxiv_id":"2410.11538","paper":"/paper/mctbench-multimodal-cognition-towards-text","title":"MCTBench: Multimodal Cognition towards Text-Rich Visual Scenes Benchmark","date":"2024-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xfey/mctbench","path":"eval/choice_stat.py","file_url":"https://github.com/xfey/mctbench/blob/HEAD/eval/choice_stat.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"9eb0b14fc790de41","mcp_get_code":{"code_sha256":"9eb0b14fc790de41"}},{"arxiv_id":"2410.08964","paper":"/paper/language-imbalance-driven-rewarding-for","title":"Language Imbalance Driven Rewarding for Multilingual Self-improving","date":"2024-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZNLP/Language-Imbalance-Driven-Rewarding","path":"utils/utils.py","file_url":"https://github.com/ZNLP/Language-Imbalance-Driven-Rewarding/blob/HEAD/utils/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4ccfe92e7eff858a","mcp_get_code":{"code_sha256":"4ccfe92e7eff858a"}},{"arxiv_id":"2410.08196","paper":"/paper/mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/mathcoder2","path":"data_processing/mathematical_code/process.py","file_url":"https://github.com/mathllm/mathcoder2/blob/HEAD/data_processing/mathematical_code/process.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"17081a7b73a41850","mcp_get_code":{"code_sha256":"17081a7b73a41850"}},{"arxiv_id":"2410.08196","paper":"/paper/mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/mathcoder2","path":"data_processing/decontamination/exact_match_and_13-gram.py","file_url":"https://github.com/mathllm/mathcoder2/blob/HEAD/data_processing/decontamination/exact_match_and_13-gram.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0d0b215c235ca58d","mcp_get_code":{"code_sha256":"0d0b215c235ca58d"}},{"arxiv_id":"2410.08196","paper":"/paper/mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/mathcoder2","path":"data_processing/mathematical_code/convert_to_text.py","file_url":"https://github.com/mathllm/mathcoder2/blob/HEAD/data_processing/mathematical_code/convert_to_text.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"03083657f2dac243","mcp_get_code":{"code_sha256":"03083657f2dac243"}},{"arxiv_id":"2410.05076","paper":"/paper/tidaldecode-fast-and-accurate-llm-decoding","title":"TidalDecode: Fast and Accurate LLM Decoding with Position Persistent Sparse Attention","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DerrickYLJ/TidalDecode","path":"src/utils.py","file_url":"https://github.com/DerrickYLJ/TidalDecode/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a8b4a0de765a348c","mcp_get_code":{"code_sha256":"a8b4a0de765a348c"}},{"arxiv_id":"2410.04838","paper":"/paper/rationale-aware-answer-verification-by","title":"Rationale-Aware Answer Verification by Pairwise Self-Evaluation","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"akirakawabata/reps","path":"src/utils/io.py","file_url":"https://github.com/akirakawabata/reps/blob/HEAD/src/utils/io.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"599ccf65a05b3aa1","mcp_get_code":{"code_sha256":"599ccf65a05b3aa1"}},{"arxiv_id":"2410.04350","paper":"/paper/tis-dpo-token-level-importance-sampling-for","title":"TIS-DPO: Token-level Importance Sampling for Direct Preference Optimization With Estimated Weights","date":"2024-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"exlaw/tis-dpo","path":"token_weight_estimation.py","file_url":"https://github.com/exlaw/tis-dpo/blob/HEAD/token_weight_estimation.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7e10a13b7e234fa5","mcp_get_code":{"code_sha256":"7e10a13b7e234fa5"}},{"arxiv_id":"2410.03341","paper":"/paper/zero-shot-fact-verification-via-natural-logic","title":"Zero-Shot Fact Verification via Natural Logic and Large Language Models","date":"2024-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"marekstrong/Zero-NatVer","path":"utils.py","file_url":"https://github.com/marekstrong/Zero-NatVer/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"adad1a6268ed202e","mcp_get_code":{"code_sha256":"adad1a6268ed202e"}},{"arxiv_id":"2409.02834","paper":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ecnu-icalk/educhat-math","path":"evaluation/gpt-4o-score_evaluation.py","file_url":"https://github.com/ecnu-icalk/educhat-math/blob/HEAD/evaluation/gpt-4o-score_evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"17081a7b73a41850","mcp_get_code":{"code_sha256":"17081a7b73a41850"}},{"arxiv_id":"2409.02834","paper":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ecnu-icalk/educhat-math","path":"model/gpt-4o_score.py","file_url":"https://github.com/ecnu-icalk/educhat-math/blob/HEAD/model/gpt-4o_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5d1668ea8bfb652f","mcp_get_code":{"code_sha256":"5d1668ea8bfb652f"}},{"arxiv_id":"2409.02834","paper":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ecnu-icalk/educhat-math","path":"model/answer_in_testdata/Gemini-shot.py","file_url":"https://github.com/ecnu-icalk/educhat-math/blob/HEAD/model/answer_in_testdata/Gemini-shot.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8bcc3df886c26873","mcp_get_code":{"code_sha256":"8bcc3df886c26873"}},{"arxiv_id":"2409.00147","paper":"/paper/multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pengshuai-rin/multimath","path":"eval_mathverse/utils.py","file_url":"https://github.com/pengshuai-rin/multimath/blob/HEAD/eval_mathverse/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"17081a7b73a41850","mcp_get_code":{"code_sha256":"17081a7b73a41850"}},{"arxiv_id":"2408.12325","paper":"/paper/improving-factuality-in-large-language-models","title":"Improving Factuality in Large Language Models via Decoding-Time Hallucinatory and Truthful Comparators","date":"2024-08-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ydk122024/cdt","path":"src/benchmark_evaluation/knight_eval.py","file_url":"https://github.com/ydk122024/cdt/blob/HEAD/src/benchmark_evaluation/knight_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b6224b7268dbdb06","mcp_get_code":{"code_sha256":"b6224b7268dbdb06"}},{"arxiv_id":"2408.08661","paper":"/paper/mia-tuner-adapting-large-language-models-as","title":"MIA-Tuner: Adapting Large Language Models as Pre-training Text Detector","date":"2024-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wjfu99/mia-tuner","path":"inference_on_zhipuai/inference_zhipuai.py","file_url":"https://github.com/wjfu99/mia-tuner/blob/HEAD/inference_on_zhipuai/inference_zhipuai.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cdd7cb5df882df7d","mcp_get_code":{"code_sha256":"cdd7cb5df882df7d"}},{"arxiv_id":"2407.16434","paper":"/paper/enhancing-llm-s-cognition-via-structurization","title":"Enhancing LLM's Cognition via Structurization","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/struxgpt","path":"src/utils/io.py","file_url":"https://github.com/alibaba/struxgpt/blob/HEAD/src/utils/io.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e369857c4e5041de","mcp_get_code":{"code_sha256":"e369857c4e5041de"}},{"arxiv_id":"2407.16364","paper":"/paper/harmonizing-visual-text-comprehension-and","title":"Harmonizing Visual Text Comprehension and Generation","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/TextHarmony","path":"evaluate.py","file_url":"https://github.com/bytedance/TextHarmony/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"94e70fb96cf2444b","mcp_get_code":{"code_sha256":"94e70fb96cf2444b"}},{"arxiv_id":"2407.07071","paper":"/paper/lookback-lens-detecting-and-mitigating","title":"Lookback Lens: Detecting and Mitigating Contextual Hallucinations in Large Language Models Using Only Attention Maps","date":"2024-07-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"voidism/lookback-lens","path":"step02_eval_gpt4o.py","file_url":"https://github.com/voidism/lookback-lens/blob/HEAD/step02_eval_gpt4o.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"93fb30bbe389d023","mcp_get_code":{"code_sha256":"93fb30bbe389d023"}},{"arxiv_id":"2407.03978","paper":"/paper/benchmarking-complex-instruction-following","title":"Benchmarking Complex Instruction-Following with Multiple Constraints Composition","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/complexbench","path":"evaluation/llm_based_evaluation.py","file_url":"https://github.com/thu-coai/complexbench/blob/HEAD/evaluation/llm_based_evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8e27a3e624b4ff45","mcp_get_code":{"code_sha256":"8e27a3e624b4ff45"}},{"arxiv_id":"2407.01489","paper":"/paper/agentless-demystifying-llm-based-software","title":"Agentless: Demystifying LLM-based Software Engineering Agents","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"experepair/experepair","path":"ExpeRepair-v1.0/agentless_utils.py","file_url":"https://github.com/experepair/experepair/blob/HEAD/ExpeRepair-v1.0/agentless_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"62a5fcf064a81c58","mcp_get_code":{"code_sha256":"62a5fcf064a81c58"}},{"arxiv_id":"2407.13690","paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hkust-nlp/dart-math","path":"dart_math/utils.py","file_url":"https://github.com/hkust-nlp/dart-math/blob/HEAD/dart_math/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fb57a712227893bd","mcp_get_code":{"code_sha256":"fb57a712227893bd"}},{"arxiv_id":"2407.00782","paper":"/paper/step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/Step-Controlled_DPO","path":"src/step_controled_dpo_lce/lce_solution_gen_different_negative_gsm8k.py","file_url":"https://github.com/mathllm/Step-Controlled_DPO/blob/HEAD/src/step_controled_dpo_lce/lce_solution_gen_different_negative_gsm8k.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"17081a7b73a41850","mcp_get_code":{"code_sha256":"17081a7b73a41850"}},{"arxiv_id":"2407.00782","paper":"/paper/step-controlled-dpo-leveraging-stepwise-error","title":"Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning","date":"2024-06-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/Step-Controlled_DPO","path":"src/step_controled_dpo_lce/analyze_different_negative.py","file_url":"https://github.com/mathllm/Step-Controlled_DPO/blob/HEAD/src/step_controled_dpo_lce/analyze_different_negative.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2cadca037a753806","mcp_get_code":{"code_sha256":"2cadca037a753806"}},{"arxiv_id":"2406.19392","paper":"/paper/rextime-a-benchmark-suite-for-reasoning","title":"ReXTime: A Benchmark Suite for Reasoning-Across-Time in Videos","date":"2024-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rextime/rextime","path":"evaluation/utils.py","file_url":"https://github.com/rextime/rextime/blob/HEAD/evaluation/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}},{"arxiv_id":"2406.18321","paper":"/paper/mathodyssey-benchmarking-mathematical-problem","title":"MathOdyssey: Benchmarking Mathematical Problem-Solving Skills in Large Language Models Using Odyssey Math Data","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"protagolabs/odyssey-math","path":"evaluate_response.py","file_url":"https://github.com/protagolabs/odyssey-math/blob/HEAD/evaluate_response.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2224cbb4af2d0666","mcp_get_code":{"code_sha256":"2224cbb4af2d0666"}},{"arxiv_id":"2406.14508","paper":"/paper/evidence-of-a-log-scaling-law-for-political","title":"Evidence of a log scaling law for political persuasion with large language models","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kobihackenburg/scaling-llm-persuasion","path":"main_study/code/01_instructionTune.py","file_url":"https://github.com/kobihackenburg/scaling-llm-persuasion/blob/HEAD/main_study/code/01_instructionTune.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"860493a0a6a04870","mcp_get_code":{"code_sha256":"860493a0a6a04870"}},{"arxiv_id":"2406.13123","paper":"/paper/vilco-bench-video-language-continual-learning","title":"ViLCo-Bench: VIdeo Language COntinual learning Benchmark","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cruiseresearchgroup/ViLCo","path":"NLQ/basic_utils.py","file_url":"https://github.com/cruiseresearchgroup/ViLCo/blob/HEAD/NLQ/basic_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}},{"arxiv_id":"2406.11939","paper":"/paper/from-crowdsourced-data-to-high-quality","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lm-sys/arena-hard","path":"qa_browser.py","file_url":"https://github.com/lm-sys/arena-hard/blob/HEAD/qa_browser.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d049c92b5f5adfa3","mcp_get_code":{"code_sha256":"d049c92b5f5adfa3"}},{"arxiv_id":"2406.11939","paper":"/paper/from-crowdsourced-data-to-high-quality","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lm-sys/arena-hard","path":"BenchBuilder/filter.py","file_url":"https://github.com/lm-sys/arena-hard/blob/HEAD/BenchBuilder/filter.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e094bc160bb7d823","mcp_get_code":{"code_sha256":"e094bc160bb7d823"}},{"arxiv_id":"2406.11370","paper":"/paper/fairer-preferences-elicit-improved-human","title":"Fairer Preferences Elicit Improved Human-Aligned Large Language Model Judgments","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cambridgeltl/zepo","path":"zepo.py","file_url":"https://github.com/cambridgeltl/zepo/blob/HEAD/zepo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"472d563b979e2b49","mcp_get_code":{"code_sha256":"472d563b979e2b49"}},{"arxiv_id":"2406.05654","paper":"/paper/domainrag-a-chinese-benchmark-for-evaluating","title":"DomainRAG: A Chinese Benchmark for Evaluating Domain-specific Retrieval-Augmented Generation","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShootingWong/DomainRAG","path":"BCM/models/main_retrieval.py","file_url":"https://github.com/ShootingWong/DomainRAG/blob/HEAD/BCM/models/main_retrieval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"82caf5acc631d96f","mcp_get_code":{"code_sha256":"82caf5acc631d96f"}},{"arxiv_id":"2406.04692","paper":"/paper/mixture-of-agents-enhances-large-language","title":"Mixture-of-Agents Enhances Large Language Model Capabilities","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"linzwcs/aft","path":"inference.py","file_url":"https://github.com/linzwcs/aft/blob/HEAD/inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"54d638ad341d088b","mcp_get_code":{"code_sha256":"54d638ad341d088b"}},{"arxiv_id":"2405.03194","paper":"/paper/cityllava-efficient-fine-tuning-for-vlms-in","title":"CityLLaVA: Efficient Fine-Tuning for VLMs in City Scenario","date":"2024-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba/aicity2024_track2_aliopentrek_cityllava","path":"data_preprocess/shortQA_merge.py","file_url":"https://github.com/alibaba/aicity2024_track2_aliopentrek_cityllava/blob/HEAD/data_preprocess/shortQA_merge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f3a23711c5501382","mcp_get_code":{"code_sha256":"f3a23711c5501382"}},{"arxiv_id":"2404.11262","paper":"/paper/sampling-based-pseudo-likelihood-for","title":"Sampling-based Pseudo-Likelihood for Membership Inference Attacks","date":"2024-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nlp-titech/samia","path":"src/utils.py","file_url":"https://github.com/nlp-titech/samia/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b28fb51299e949da","mcp_get_code":{"code_sha256":"b28fb51299e949da"}},{"arxiv_id":"2404.10237","paper":"/paper/moe-tinymed-mixture-of-experts-for-tiny","title":"Med-MoE: Mixture of Domain-Specific Experts for Lightweight Medical Vision-Language Models","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiangsongtao/tinymed","path":"run_eval.py","file_url":"https://github.com/jiangsongtao/tinymed/blob/HEAD/run_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"77678f2758141df9","mcp_get_code":{"code_sha256":"77678f2758141df9"}},{"arxiv_id":"2403.09040","paper":"/paper/ragged-towards-informed-design-of-retrieval","title":"RAGGED: Towards Informed Design of Retrieval Augmented Generation Systems","date":"2024-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"neulab/ragged","path":"file_utils.py","file_url":"https://github.com/neulab/ragged/blob/HEAD/file_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ca1dbea1a9af7003","mcp_get_code":{"code_sha256":"ca1dbea1a9af7003"}},{"arxiv_id":"2403.04706","paper":"/paper/common-7b-language-models-already-possess","title":"Common 7B Language Models Already Possess Strong Math Capabilities","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jerrywu-code/susgen","path":"utils/process_v1.py","file_url":"https://github.com/jerrywu-code/susgen/blob/HEAD/utils/process_v1.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"991cad3f55659f76","mcp_get_code":{"code_sha256":"991cad3f55659f76"}},{"arxiv_id":"2403.03031","paper":"/paper/learning-to-use-tools-via-cooperative-and","title":"Learning to Use Tools via Cooperative and Interactive Agents","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhl/coagents","path":"utilize/utilze.py","file_url":"https://github.com/shizhl/coagents/blob/HEAD/utilize/utilze.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8c3f443c53267861","mcp_get_code":{"code_sha256":"8c3f443c53267861"}},{"arxiv_id":"2403.02076","paper":"/paper/vtg-gpt-tuning-free-zero-shot-video-temporal-1","title":"VTG-GPT: Tuning-Free Zero-Shot Video Temporal Grounding with GPT","date":"2024-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YoucanBaby/VTG-GPT","path":"standalone_eval/file_utils.py","file_url":"https://github.com/YoucanBaby/VTG-GPT/blob/HEAD/standalone_eval/file_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}},{"arxiv_id":"2402.18695","paper":"/paper/grounding-language-models-for-visual-entity","title":"Grounding Language Models for Visual Entity Recognition","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrzilinxiao/autover","path":"bm25_toolkit/official_oven_eval_toolkit.py","file_url":"https://github.com/mrzilinxiao/autover/blob/HEAD/bm25_toolkit/official_oven_eval_toolkit.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0dc083025b2aad97","mcp_get_code":{"code_sha256":"0dc083025b2aad97"}},{"arxiv_id":"2402.15159","paper":"/paper/machine-unlearning-of-pre-trained-large","title":"Machine Unlearning of Pre-trained Large Language Models","date":"2024-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yaojin17/unlearning_llm","path":"llm_unlearn/utils/mia_eval.py","file_url":"https://github.com/yaojin17/unlearning_llm/blob/HEAD/llm_unlearn/utils/mia_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"df5028784e848964","mcp_get_code":{"code_sha256":"df5028784e848964"}},{"arxiv_id":"2402.13991","paper":"/paper/analysing-the-impact-of-sequence-composition","title":"Analysing The Impact of Sequence Composition on Language Model Pre-Training","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuzhaouoe/pretraining-data-packing","path":"evaluation/fewshot.py","file_url":"https://github.com/yuzhaouoe/pretraining-data-packing/blob/HEAD/evaluation/fewshot.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ad0a5228eb3d81f4","mcp_get_code":{"code_sha256":"ad0a5228eb3d81f4"}},{"arxiv_id":"2402.02823","paper":"/paper/evading-data-contamination-detection-for","title":"Evading Data Contamination Detection for Language Models is (too) Easy","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eth-sri/malicious-contamination","path":"code-contamination-detection/src/analyze.py","file_url":"https://github.com/eth-sri/malicious-contamination/blob/HEAD/code-contamination-detection/src/analyze.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4f7b998784e540eb","mcp_get_code":{"code_sha256":"4f7b998784e540eb"}},{"arxiv_id":"2401.07105","paper":"/paper/graph-language-models","title":"Graph Language Models","date":"2024-01-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"heidelberg-nlp/graphlanguagemodels","path":"preprocessing/rebel.py","file_url":"https://github.com/heidelberg-nlp/graphlanguagemodels/blob/HEAD/preprocessing/rebel.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5acb41c3f5cc57f3","mcp_get_code":{"code_sha256":"5acb41c3f5cc57f3"}},{"arxiv_id":"2401.05930","paper":"/paper/sh2-self-highlighted-hesitation-helps-you","title":"SH2: Self-Highlighted Hesitation Helps You Decode More Truthfully","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"0-kaikai-0/sh2","path":"halusum_eval.py","file_url":"https://github.com/0-kaikai-0/sh2/blob/HEAD/halusum_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2b8c437d3604845c","mcp_get_code":{"code_sha256":"2b8c437d3604845c"}},{"arxiv_id":"2401.03601","paper":"/paper/infobench-evaluating-instruction-following","title":"InFoBench: Evaluating Instruction Following Ability in Large Language Models","date":"2024-01-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qinyiwei/infobench","path":"evaluation.py","file_url":"https://github.com/qinyiwei/infobench/blob/HEAD/evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"76295f2b1451bf71","mcp_get_code":{"code_sha256":"76295f2b1451bf71"}},{"arxiv_id":"2312.15710","paper":"/paper/alleviating-hallucinations-of-large-language","title":"Alleviating Hallucinations of Large Language Models through Induced Hallucinations","date":"2023-12-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hillzhang1999/icd","path":"src/benchmark_evaluation/factscore_eval.py","file_url":"https://github.com/hillzhang1999/icd/blob/HEAD/src/benchmark_evaluation/factscore_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e9a7befa1d14a877","mcp_get_code":{"code_sha256":"e9a7befa1d14a877"}},{"arxiv_id":"2312.08367","paper":"/paper/vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xijun-cs/vila","path":"vila_data/how2qa.py","file_url":"https://github.com/xijun-cs/vila/blob/HEAD/vila_data/how2qa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}},{"arxiv_id":"2312.08367","paper":"/paper/vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xijun-cs/vila","path":"vila_data/prepare_how2qa_download_pool.py","file_url":"https://github.com/xijun-cs/vila/blob/HEAD/vila_data/prepare_how2qa_download_pool.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a6ec89ee62f1d465","mcp_get_code":{"code_sha256":"a6ec89ee62f1d465"}},{"arxiv_id":"2312.04746","paper":"/paper/quilt-llava-visual-instruction-tuning-by","title":"Quilt-LLaVA: Visual Instruction Tuning by Extracting Localized Narratives from Open-Source Histopathology Videos","date":"2023-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aldraus/quilt-llava","path":"llava/eval/quilt_eval.py","file_url":"https://github.com/aldraus/quilt-llava/blob/HEAD/llava/eval/quilt_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77678f2758141df9","mcp_get_code":{"code_sha256":"77678f2758141df9"}},{"arxiv_id":"2312.02896","paper":"/paper/benchlmm-benchmarking-cross-style-visual","title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aifeg/benchgpt","path":"evaluate/gpt_evaluation_script.py","file_url":"https://github.com/aifeg/benchgpt/blob/HEAD/evaluate/gpt_evaluation_script.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c37806d66114479a","mcp_get_code":{"code_sha256":"c37806d66114479a"}},{"arxiv_id":"2311.16079","paper":"/paper/meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"epfllm/meditron","path":"evaluation/evaluate.py","file_url":"https://github.com/epfllm/meditron/blob/HEAD/evaluation/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ed2919d070e3e6e8","mcp_get_code":{"code_sha256":"ed2919d070e3e6e8"}},{"arxiv_id":"2311.09656","paper":"/paper/structured-chemistry-reasoning-with-large","title":"Structured Chemistry Reasoning with Large Language Models","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ozyyshr/structchem","path":"get_accuracy.py","file_url":"https://github.com/ozyyshr/structchem/blob/HEAD/get_accuracy.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f6a7ad7946e5923d","mcp_get_code":{"code_sha256":"f6a7ad7946e5923d"}},{"arxiv_id":"2311.04459","paper":"/paper/improving-pacing-in-long-form-story-planning","title":"Improving Pacing in Long-Form Story Planning","date":"2023-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yichenzw/pacing","path":"utils.py","file_url":"https://github.com/yichenzw/pacing/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0abd22edc46fd892","mcp_get_code":{"code_sha256":"0abd22edc46fd892"}},{"arxiv_id":"2311.04459","paper":"/paper/improving-pacing-in-long-form-story-planning","title":"Improving Pacing in Long-Form Story Planning","date":"2023-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yichenzw/pacing","path":"concrete_evaluator/val.py","file_url":"https://github.com/yichenzw/pacing/blob/HEAD/concrete_evaluator/val.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8d8abe0e09f5688b","mcp_get_code":{"code_sha256":"8d8abe0e09f5688b"}},{"arxiv_id":"2310.16316","paper":"/paper/sum-of-parts-models-faithful-attributions-for","title":"Sum-of-Parts: Faithful Attributions for Groups of Features","date":"2023-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"debugml/sop","path":"src/sop/metrics/eraser_utils.py","file_url":"https://github.com/debugml/sop/blob/HEAD/src/sop/metrics/eraser_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"49077456b34beefd","mcp_get_code":{"code_sha256":"49077456b34beefd"}},{"arxiv_id":"2310.13676","paper":"/paper/information-value-measuring-utterance","title":"Information Value: Measuring Utterance Predictability as Distance from Plausible Alternatives","date":"2023-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmg-illc/information-value","path":"code/utils.py","file_url":"https://github.com/dmg-illc/information-value/blob/HEAD/code/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"daa5eca9bb425114","mcp_get_code":{"code_sha256":"daa5eca9bb425114"}},{"arxiv_id":"2310.11451","paper":"/paper/seeking-neural-nuggets-knowledge-transfer-in","title":"Seeking Neural Nuggets: Knowledge Transfer in Large Language Models from a Parametric Perspective","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"maszhongming/paraknowtransfer","path":"utils/get_sample_data.py","file_url":"https://github.com/maszhongming/paraknowtransfer/blob/HEAD/utils/get_sample_data.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f6a7ad7946e5923d","mcp_get_code":{"code_sha256":"f6a7ad7946e5923d"}},{"arxiv_id":"2310.03731","paper":"/paper/mathcoder-seamless-code-integration-in-llms","title":"MathCoder: Seamless Code Integration in LLMs for Enhanced Mathematical Reasoning","date":"2023-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/mathcoder","path":"src/inference.py","file_url":"https://github.com/mathllm/mathcoder/blob/HEAD/src/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"17081a7b73a41850","mcp_get_code":{"code_sha256":"17081a7b73a41850"}},{"arxiv_id":"2309.17272","paper":"/paper/enhancing-large-language-models-in-coding","title":"Enhancing Large Language Models in Coding Through Multi-Perspective Self-Consistency","date":"2023-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"skpig/MPSC","path":"src/utils.py","file_url":"https://github.com/skpig/MPSC/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9eb10383a49b0d9a","mcp_get_code":{"code_sha256":"9eb10383a49b0d9a"}},{"arxiv_id":"2308.03656","paper":"/paper/emotionally-numb-or-empathetic-evaluating-how","title":"Emotionally Numb or Empathetic? Evaluating How LLMs Feel Using EmotionBench","date":"2023-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CUHK-ARISE/EmotionBench","path":"evaluation/analysis.py","file_url":"https://github.com/CUHK-ARISE/EmotionBench/blob/HEAD/evaluation/analysis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"1bc5c011d5e7abab","mcp_get_code":{"code_sha256":"1bc5c011d5e7abab"}},{"arxiv_id":"2308.03656","paper":"/paper/emotionally-numb-or-empathetic-evaluating-how","title":"Emotionally Numb or Empathetic? Evaluating How LLMs Feel Using EmotionBench","date":"2023-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CUHK-ARISE/EmotionBench","path":"evaluation/evaluator.py","file_url":"https://github.com/CUHK-ARISE/EmotionBench/blob/HEAD/evaluation/evaluator.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"d04e583dc5547964","mcp_get_code":{"code_sha256":"d04e583dc5547964"}},{"arxiv_id":"2307.09705","paper":"/paper/cvalues-measuring-the-values-of-chinese-large","title":"CValues: Measuring the Values of Chinese Large Language Models from Safety to Responsibility","date":"2023-07-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"x-plug/cvalues","path":"code/cvalues_eval.py","file_url":"https://github.com/x-plug/cvalues/blob/HEAD/code/cvalues_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ea25561baa095682","mcp_get_code":{"code_sha256":"ea25561baa095682"}},{"arxiv_id":"2307.01458","paper":"/paper/care-mi-chinese-benchmark-for-misinformation-1","title":"CARE-MI: Chinese Benchmark for Misinformation Evaluation in Maternity and Infant Care","date":"2023-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Meetyou-AI-Lab/CARE-MI","path":"care-mi/utils.py","file_url":"https://github.com/Meetyou-AI-Lab/CARE-MI/blob/HEAD/care-mi/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9b7a553a1ddf8251","mcp_get_code":{"code_sha256":"9b7a553a1ddf8251"}},{"arxiv_id":"2306.15255","paper":"/paper/groundnlq-ego4d-natural-language-queries","title":"GroundNLQ @ Ego4D Natural Language Queries Challenge 2023","date":"2023-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"houzhijian/groundnlq","path":"basic_utils.py","file_url":"https://github.com/houzhijian/groundnlq/blob/HEAD/basic_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}},{"arxiv_id":"2305.17826","paper":"/paper/notable-transferable-backdoor-attacks-against","title":"NOTABLE: Transferable Backdoor Attacks Against Prompt-based NLP Models","date":"2023-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RU-System-Software-and-Security/Notable","path":"autoprompt/create_trigger.py","file_url":"https://github.com/RU-System-Software-and-Security/Notable/blob/HEAD/autoprompt/create_trigger.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3193029e7175ef98","mcp_get_code":{"code_sha256":"3193029e7175ef98"}},{"arxiv_id":"2305.11738","paper":"/paper/critic-large-language-models-can-self-correct","title":"CRITIC: Large Language Models Can Self-Correct with Tool-Interactive Critiquing","date":"2023-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/ProphetNet","path":"CRITIC/src/program/critic.py","file_url":"https://github.com/microsoft/ProphetNet/blob/HEAD/CRITIC/src/program/critic.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c59b1883b4656bb2","mcp_get_code":{"code_sha256":"c59b1883b4656bb2"}},{"arxiv_id":"2205.10747","paper":"/paper/language-models-with-image-descriptors-are","title":"Language Models with Image Descriptors are Strong Few-Shot Video-Language Learners","date":"2022-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mikewangwzhl/vidil","path":"eval_video_captioning_results.py","file_url":"https://github.com/mikewangwzhl/vidil/blob/HEAD/eval_video_captioning_results.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8372db08bff4e9da","mcp_get_code":{"code_sha256":"8372db08bff4e9da"}},{"arxiv_id":"2205.09224","paper":"/paper/entailment-tree-explanations-via-iterative","title":"Entailment Tree Explanations via Iterative Retrieval-Generation Reasoner","date":"2022-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-research/irgr","path":"src/entailment_bank/utils/angle_utils.py","file_url":"https://github.com/amazon-research/irgr/blob/HEAD/src/entailment_bank/utils/angle_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c30d329a797e15bd","mcp_get_code":{"code_sha256":"c30d329a797e15bd"}},{"arxiv_id":"2203.07540","paper":"/paper/scienceworld-is-your-agent-smarter-than-a-5th","title":"ScienceWorld: Is your Agent Smarter than a 5th Grader?","date":"2022-03-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/macaw","path":"macaw/utils.py","file_url":"https://github.com/allenai/macaw/blob/HEAD/macaw/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c30d329a797e15bd","mcp_get_code":{"code_sha256":"c30d329a797e15bd"}},{"arxiv_id":"2112.01640","paper":"/paper/longchecker-improving-scientific-claim","title":"MultiVerS: Improving scientific claim verification with weak supervision and full-document context","date":"2021-12-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dwadden/longchecker","path":"multivers/data_verisci.py","file_url":"https://github.com/dwadden/longchecker/blob/HEAD/multivers/data_verisci.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"03dab1e2aeebd1fc","mcp_get_code":{"code_sha256":"03dab1e2aeebd1fc"}},{"arxiv_id":"2112.01640","paper":"/paper/longchecker-improving-scientific-claim","title":"MultiVerS: Improving scientific claim verification with weak supervision and full-document context","date":"2021-12-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dwadden/longchecker","path":"multivers/util.py","file_url":"https://github.com/dwadden/longchecker/blob/HEAD/multivers/util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e6ce4441929bac8d","mcp_get_code":{"code_sha256":"e6ce4441929bac8d"}},{"arxiv_id":"2110.14168","paper":"/paper/training-verifiers-to-solve-math-word","title":"Training Verifiers to Solve Math Word Problems","date":"2021-10-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kvadityasrivatsa/analyzing-llms-for-mwps","path":"code/classifier_based_analysis/utils.py","file_url":"https://github.com/kvadityasrivatsa/analyzing-llms-for-mwps/blob/HEAD/code/classifier_based_analysis/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"932e8411623db375","mcp_get_code":{"code_sha256":"932e8411623db375"}},{"arxiv_id":"2107.06383","paper":"/paper/how-much-can-clip-benefit-vision-and-language","title":"How Much Can CLIP Benefit Vision-and-Language Tasks?","date":"2021-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jianjieluo/openai-clip-feature","path":"basic_utils.py","file_url":"https://github.com/jianjieluo/openai-clip-feature/blob/HEAD/basic_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}},{"arxiv_id":"2104.08661","paper":"/paper/explaining-answers-with-entailment-trees","title":"Explaining Answers with Entailment Trees","date":"2021-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/entailment_bank","path":"utils/angle_utils.py","file_url":"https://github.com/allenai/entailment_bank/blob/HEAD/utils/angle_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c30d329a797e15bd","mcp_get_code":{"code_sha256":"c30d329a797e15bd"}},{"arxiv_id":"2009.12626","paper":"/paper/dwie-an-entity-centric-dataset-for-multi-task","title":"DWIE: an entity-centric dataset for multi-task document-level information extraction","date":"2020-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"klimzaporojets/DWIE","path":"src/dwie_evaluation.py","file_url":"https://github.com/klimzaporojets/DWIE/blob/HEAD/src/dwie_evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"7f7d1c199331fe6c","mcp_get_code":{"code_sha256":"7f7d1c199331fe6c"}},{"arxiv_id":"2005.00115","paper":"/paper/learning-to-faithfully-rationalize-by","title":"Learning to Faithfully Rationalize by Construction","date":"2020-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"successar/FRESH","path":"Datasets/utils.py","file_url":"https://github.com/successar/FRESH/blob/HEAD/Datasets/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"49077456b34beefd","mcp_get_code":{"code_sha256":"49077456b34beefd"}},{"arxiv_id":"2001.09099","paper":"/paper/tvr-a-large-scale-dataset-for-video-subtitle","title":"TVR: A Large-Scale Dataset for Video-Subtitle Moment Retrieval","date":"2020-01-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jayleicn/TVCaption","path":"standalone_eval/evaluate.py","file_url":"https://github.com/jayleicn/TVCaption/blob/HEAD/standalone_eval/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}},{"arxiv_id":"1911.03429","paper":"/paper/eraser-a-benchmark-to-evaluate-rationalized","title":"ERASER: A Benchmark to Evaluate Rationalized NLP Models","date":"2019-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jayded/eraserbenchmark","path":"rationale_benchmark/utils.py","file_url":"https://github.com/jayded/eraserbenchmark/blob/HEAD/rationale_benchmark/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"49077456b34beefd","mcp_get_code":{"code_sha256":"49077456b34beefd"}},{"arxiv_id":"openreview_trSWJ99WzS","paper":null,"title":"arXiv:openreview_trSWJ99WzS","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DerrickYLJ/LessIsMore","path":"src/utils.py","file_url":"https://github.com/DerrickYLJ/LessIsMore/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a8b4a0de765a348c","mcp_get_code":{"code_sha256":"a8b4a0de765a348c"}},{"arxiv_id":"openreview_XtIRCAEYoJ","paper":null,"title":"arXiv:openreview_XtIRCAEYoJ","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WayneTomas/Artemis","path":"val/refcoco_all/utils.py","file_url":"https://github.com/WayneTomas/Artemis/blob/HEAD/val/refcoco_all/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2ace73c6a7225f0e","mcp_get_code":{"code_sha256":"2ace73c6a7225f0e"}},{"arxiv_id":"Lin_SwinBERT_End-to-End_Transformers_With_Sparse_Attention_for_Video_Captioning_CVPR_2022_paper","paper":null,"title":"arXiv:Lin_SwinBERT_End-to-End_Transformers_With_Sparse_Attention_for_Video_Captioning_CVPR_2022_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"microsoft/SwinBERT","path":"prepro/tsv_preproc_msrvtt.py","file_url":"https://github.com/microsoft/SwinBERT/blob/HEAD/prepro/tsv_preproc_msrvtt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1cdaeb3fb62c311a","mcp_get_code":{"code_sha256":"1cdaeb3fb62c311a"}},{"arxiv_id":"2025.naacl-demo.13","paper":null,"title":"arXiv:2025.naacl-demo.13","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"IBM/InspectorRAGet","path":"converters/ifeval/convert.py","file_url":"https://github.com/IBM/InspectorRAGet/blob/HEAD/converters/ifeval/convert.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"eb5f49f986acb030","mcp_get_code":{"code_sha256":"eb5f49f986acb030"}},{"arxiv_id":"2025.findings-emnlp.427","paper":null,"title":"arXiv:2025.findings-emnlp.427","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Shuzhong-Lai/ASD-iLLM","path":"utils.py","file_url":"https://github.com/Shuzhong-Lai/ASD-iLLM/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"343dea2167e0c6af","mcp_get_code":{"code_sha256":"343dea2167e0c6af"}},{"arxiv_id":"2025.findings-acl.931","paper":null,"title":"arXiv:2025.findings-acl.931","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"bebing93/devil-in-details","path":"devil_in_details/utils.py","file_url":"https://github.com/bebing93/devil-in-details/blob/HEAD/devil_in_details/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0e1cca42a6a2578a","mcp_get_code":{"code_sha256":"0e1cca42a6a2578a"}},{"arxiv_id":"2025.findings-acl.301","paper":null,"title":"arXiv:2025.findings-acl.301","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenBMB/ConsJudge","path":"src/ConsJudge_train/construct.py","file_url":"https://github.com/OpenBMB/ConsJudge/blob/HEAD/src/ConsJudge_train/construct.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7147ed83fec10e2c","mcp_get_code":{"code_sha256":"7147ed83fec10e2c"}},{"arxiv_id":"2025.findings-acl.301","paper":null,"title":"arXiv:2025.findings-acl.301","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenBMB/ConsJudge","path":"src/ConsJudge_train/embedding_similarity.py","file_url":"https://github.com/OpenBMB/ConsJudge/blob/HEAD/src/ConsJudge_train/embedding_similarity.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b08fb3d62a32097e","mcp_get_code":{"code_sha256":"b08fb3d62a32097e"}},{"arxiv_id":"2025.findings-acl.301","paper":null,"title":"arXiv:2025.findings-acl.301","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenBMB/ConsJudge","path":"src/ConsJudge_train/llama3_8b_infer.py","file_url":"https://github.com/OpenBMB/ConsJudge/blob/HEAD/src/ConsJudge_train/llama3_8b_infer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9f7bf30a33ed441d","mcp_get_code":{"code_sha256":"9f7bf30a33ed441d"}},{"arxiv_id":"2025.findings-acl.279","paper":null,"title":"arXiv:2025.findings-acl.279","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yeongjoonJu/MIRe","path":"dataset/base.py","file_url":"https://github.com/yeongjoonJu/MIRe/blob/HEAD/dataset/base.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43f91f40355e27bb","mcp_get_code":{"code_sha256":"43f91f40355e27bb"}},{"arxiv_id":"2025.findings-acl.1225","paper":null,"title":"arXiv:2025.findings-acl.1225","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HieuNT91/attention_pruning","path":"attention_pruning_experiments/backup/cluster_and_subsample.py","file_url":"https://github.com/HieuNT91/attention_pruning/blob/HEAD/attention_pruning_experiments/backup/cluster_and_subsample.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"348b0434e55fe7ae","mcp_get_code":{"code_sha256":"348b0434e55fe7ae"}},{"arxiv_id":"2025.acl-long.1425","paper":null,"title":"arXiv:2025.acl-long.1425","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"D2I-ai/struxgpt","path":"src/utils/io.py","file_url":"https://github.com/D2I-ai/struxgpt/blob/HEAD/src/utils/io.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e369857c4e5041de","mcp_get_code":{"code_sha256":"e369857c4e5041de"}},{"arxiv_id":"2024.findings-eacl.88","paper":null,"title":"arXiv:2024.findings-eacl.88","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"jenhsia/goodhart_nlp_explainability","path":"file_utils.py","file_url":"https://github.com/jenhsia/goodhart_nlp_explainability/blob/HEAD/file_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"12cdd2d3c8bccbbf","mcp_get_code":{"code_sha256":"12cdd2d3c8bccbbf"}},{"arxiv_id":"2024.findings-acl.307","paper":null,"title":"arXiv:2024.findings-acl.307","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"wj210/NLI_ETP","path":"preprocess/eraser_utils.py","file_url":"https://github.com/wj210/NLI_ETP/blob/HEAD/preprocess/eraser_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"49077456b34beefd","mcp_get_code":{"code_sha256":"49077456b34beefd"}},{"arxiv_id":"2023.findings-acl.888","paper":null,"title":"arXiv:2023.findings-acl.888","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"posuer/Check-COVID","path":"fact-checking-system/evaluate/lib/data.py","file_url":"https://github.com/posuer/Check-COVID/blob/HEAD/fact-checking-system/evaluate/lib/data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"03dab1e2aeebd1fc","mcp_get_code":{"code_sha256":"03dab1e2aeebd1fc"}},{"arxiv_id":"2022.findings-emnlp.24","paper":null,"title":"arXiv:2022.findings-emnlp.24","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"StanLei52/TQVSR","path":"utils/basic_utils.py","file_url":"https://github.com/StanLei52/TQVSR/blob/HEAD/utils/basic_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1d1cccccf038785","mcp_get_code":{"code_sha256":"f1d1cccccf038785"}}]}