{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extract-answer","entry":"extract_answer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":112,"n_papers_ran":68,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":112,"n_samples_ran":66,"n_samples_fingerprinted":59,"n_places":126,"n_places_pointer_only":55,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":26,"ran_fixture":0,"ran":39,"unverified":46},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.12265","paper":"/paper/arxiv-2609-12265","title":"GTA: Graph Theory Agent and Benchmark for Algorithmic Graph Reasoning with LLMs","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"xzx34/GTA","path":"src/gtbench/evaluate.py","file_url":"https://github.com/xzx34/GTA/blob/HEAD/src/gtbench/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0a5cb941be6403a6","mcp_get_code":{"code_sha256":"0a5cb941be6403a6"}},{"arxiv_id":"2609.12243","paper":"/paper/arxiv-2609-12243","title":"Chopthin-Consensus Power Sampling: A Diversity-Preserving Approach to LLM Decoding","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"MinooAhmadii/chopthin-consensus-power-sampling","path":"ccps/graders/answers.py","file_url":"https://github.com/MinooAhmadii/chopthin-consensus-power-sampling/blob/HEAD/ccps/graders/answers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"31f2eef78b769944","mcp_get_code":{"code_sha256":"31f2eef78b769944"}},{"arxiv_id":"2608.17809","paper":"/paper/arxiv-2608-17809","title":"Whether LLMs Can Navigate Beliefs and Facts Depends on How You Phrase It","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ngqm/belief-fact-phrasing","path":"src/eval/instruction_effect_table.py","file_url":"https://github.com/ngqm/belief-fact-phrasing/blob/HEAD/src/eval/instruction_effect_table.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"520d0858be37dec1","mcp_get_code":{"code_sha256":"520d0858be37dec1"}},{"arxiv_id":"2608.05732","paper":"/paper/arxiv-2608-05732","title":"CircuitSteer: Geometrically Aligned Multi-Layer Steering via Sparse Autoencoder Circuits","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"mehrshad-sdtn/CircuitSteer","path":"experiments/gsm8k.py","file_url":"https://github.com/mehrshad-sdtn/CircuitSteer/blob/HEAD/experiments/gsm8k.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"038ceb4e56fe6031","mcp_get_code":{"code_sha256":"038ceb4e56fe6031"}},{"arxiv_id":"2607.26627","paper":"/paper/arxiv-2607-26627","title":"Revisiting Lossy Verification in Speculative Decoding: Mechanisms, Trade-offs, and Failure Modes","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"ZhouYuxuanYX/Fast-HSD","path":"fast_hsd/benchmarks/_math_scoring.py","file_url":"https://github.com/ZhouYuxuanYX/Fast-HSD/blob/HEAD/fast_hsd/benchmarks/_math_scoring.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"69e80a64efad173b","mcp_get_code":{"code_sha256":"69e80a64efad173b"}},{"arxiv_id":"2606.16154","paper":"/paper/arxiv-2606-16154","title":"A Gradient Perspective on RLVR Stability and Winner Advantage Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"layer6ai-labs/wapo","path":"environments/math_dapo/math_dapo.py","file_url":"https://github.com/layer6ai-labs/wapo/blob/HEAD/environments/math_dapo/math_dapo.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f3be998c40770890","mcp_get_code":{"code_sha256":"f3be998c40770890"}},{"arxiv_id":"2606.16154","paper":"/paper/arxiv-2606-16154","title":"A Gradient Perspective on RLVR Stability and Winner Advantage Policy Optimization","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"layer6ai-labs/wapo","path":"environments/math_prm/math_prm.py","file_url":"https://github.com/layer6ai-labs/wapo/blob/HEAD/environments/math_prm/math_prm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"059412997faa2f93","mcp_get_code":{"code_sha256":"059412997faa2f93"}},{"arxiv_id":"2606.00579","paper":"/paper/arxiv-2606-00579","title":"Sandboxed Coding Agents are Competitive Omni-modal Task Solvers","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Dongping-Chen/OmniCoding","path":"src/omnicoding/rl/reward.py","file_url":"https://github.com/Dongping-Chen/OmniCoding/blob/HEAD/src/omnicoding/rl/reward.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"51e9b7e8d7ad73f8","mcp_get_code":{"code_sha256":"51e9b7e8d7ad73f8"}},{"arxiv_id":"2605.26952","paper":"/paper/arxiv-2605-26952","title":"Efficient Agentic Reinforcement Learning with On-Policy Intrinsic Knowledge Boundary Enhancement","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"CuSO4-Chen/AKBE","path":"AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","file_url":"https://github.com/CuSO4-Chen/AKBE/blob/HEAD/AKBE/verl_akbe/verl/utils/reward_score/reward_em_betagrpo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8aec27dfaeff217d","mcp_get_code":{"code_sha256":"8aec27dfaeff217d"}},{"arxiv_id":"2605.26414","paper":"/paper/arxiv-2605-26414","title":"Reasoning, Code, or Both? How Large Language Models Handle Variations in Math Questions","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"masamodelkin/llm-robustness-code-execution","path":"src/evals/CoT.py","file_url":"https://github.com/masamodelkin/llm-robustness-code-execution/blob/HEAD/src/evals/CoT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f8f102d5cfc1883c","mcp_get_code":{"code_sha256":"f8f102d5cfc1883c"}},{"arxiv_id":"2605.25920","paper":"/paper/arxiv-2605-25920","title":"Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"AlexFanw/LegalSearch-R1","path":"user/calculate_metrics.py","file_url":"https://github.com/AlexFanw/LegalSearch-R1/blob/HEAD/user/calculate_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"04f6d2711d4cd42e","mcp_get_code":{"code_sha256":"04f6d2711d4cd42e"}},{"arxiv_id":"2605.23940","paper":"/paper/arxiv-2605-23940","title":"Published at ICLR 2026 Workshop on Reasoning and Planning for LLMs RESIDUAL DRIFT DOMINATES CONTRADICTION IN MULTI-TURN CONSTRAINT REASONING","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"kaons-research/drift-bench","path":"src/extraction.py","file_url":"https://github.com/kaons-research/drift-bench/blob/HEAD/src/extraction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2859b0d4274869b4","mcp_get_code":{"code_sha256":"2859b0d4274869b4"}},{"arxiv_id":"2605.17672","paper":"/paper/arxiv-2605-17672","title":"Stop When Reasoning Converges: Semantic-Preserving Early Exit for Reasoning Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"giovanni-vaccarino/PUMA","path":"puma/extract_final_candidates.py","file_url":"https://github.com/giovanni-vaccarino/PUMA/blob/HEAD/puma/extract_final_candidates.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"000a843cb1f68ddb","mcp_get_code":{"code_sha256":"000a843cb1f68ddb"}},{"arxiv_id":"2605.06623","paper":"/paper/arxiv-2605-06623","title":"MASPO: Joint Prompt Optimization for LLM-based Multi-Agent Systems","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"wangzx1219/MASPO","path":"utils.py","file_url":"https://github.com/wangzx1219/MASPO/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8ae4dd0f7d0bba2f","mcp_get_code":{"code_sha256":"8ae4dd0f7d0bba2f"}},{"arxiv_id":"2605.05973","paper":"/paper/arxiv-2605-05973","title":"Towards Reliable LLM Evaluation: Correcting the Winner's Curse in Adaptive Benchmarking","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"jznmsl/siren","path":"03_analysis_scripts/study_e_multimodel_api.py","file_url":"https://github.com/jznmsl/siren/blob/HEAD/03_analysis_scripts/study_e_multimodel_api.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"99767d2dbf24d8cf","mcp_get_code":{"code_sha256":"99767d2dbf24d8cf"}},{"arxiv_id":"2605.05973","paper":"/paper/arxiv-2605-05973","title":"Towards Reliable LLM Evaluation: Correcting the Winner's Curse in Adaptive Benchmarking","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"jznmsl/siren","path":"03_analysis_scripts/study_e_multimodel_law.py","file_url":"https://github.com/jznmsl/siren/blob/HEAD/03_analysis_scripts/study_e_multimodel_law.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"16e4d0149abd853f","mcp_get_code":{"code_sha256":"16e4d0149abd853f"}},{"arxiv_id":"2605.05973","paper":"/paper/arxiv-2605-05973","title":"Towards Reliable LLM Evaluation: Correcting the Winner's Curse in Adaptive Benchmarking","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"jznmsl/siren","path":"03_analysis_scripts/build_tensor.py","file_url":"https://github.com/jznmsl/siren/blob/HEAD/03_analysis_scripts/build_tensor.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bec005b22dd07b60","mcp_get_code":{"code_sha256":"bec005b22dd07b60"}},{"arxiv_id":"2605.03460","paper":"/paper/arxiv-2605-03460","title":"FinSTaR: Towards Financial Reasoning with Time Series Reasoning Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"seunghan96/FinSTaR","path":"src/evaluation/utils.py","file_url":"https://github.com/seunghan96/FinSTaR/blob/HEAD/src/evaluation/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ed37ec1880683750","mcp_get_code":{"code_sha256":"ed37ec1880683750"}},{"arxiv_id":"2604.16256","paper":"/paper/arxiv-2604-16256","title":"Do Vision-Language Models Truly Perform Vision Reasoning? A Rigorous Study of the Modality Gap","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"xuyige/CrossMath","path":"batch_inference_qwen35.py","file_url":"https://github.com/xuyige/CrossMath/blob/HEAD/batch_inference_qwen35.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"c53d52049bcc871f","mcp_get_code":{"code_sha256":"c53d52049bcc871f"}},{"arxiv_id":"2604.13731","paper":"/paper/arxiv-2604-13731","title":"Doc-V * : Coarse-to-Fine Interactive Visual Reasoning for Multi-Page Document VQA","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"SeerRay-Lab/Doc-V","path":"inference/agent.py","file_url":"https://github.com/SeerRay-Lab/Doc-V/blob/HEAD/inference/agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"52a7f587ad7bd317","mcp_get_code":{"code_sha256":"52a7f587ad7bd317"}},{"arxiv_id":"2604.13731","paper":"/paper/arxiv-2604-13731","title":"Doc-V * : Coarse-to-Fine Interactive Visual Reasoning for Multi-Page Document VQA","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"SeerRay-Lab/Doc-V","path":"inference/rag.py","file_url":"https://github.com/SeerRay-Lab/Doc-V/blob/HEAD/inference/rag.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2dcb4cc46acab4c7","mcp_get_code":{"code_sha256":"2dcb4cc46acab4c7"}},{"arxiv_id":"2604.11841","paper":"/paper/arxiv-2604-11841","title":"Polynomial Expansion Rank Adaptation: Enhancing Low-Rank Fine-Tuning with High-Order Interactions","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zhangwenhao6/PERA","path":"eval_commonsense.py","file_url":"https://github.com/zhangwenhao6/PERA/blob/HEAD/eval_commonsense.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"d14efab97e468115","mcp_get_code":{"code_sha256":"d14efab97e468115"}},{"arxiv_id":"2604.11048","paper":"/paper/arxiv-2604-11048","title":"A Systematic Analysis of the Impact of Persona Steering on LLM Capabilities","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"cjia7/DPR","path":"src/npti/eval/eval_mmlu.py","file_url":"https://github.com/cjia7/DPR/blob/HEAD/src/npti/eval/eval_mmlu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ad16da4e4f52ed7","mcp_get_code":{"code_sha256":"0ad16da4e4f52ed7"}},{"arxiv_id":"2604.03616","paper":"/paper/arxiv-2604-03616","title":"The Format Tax","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ivnle/the-format-tax","path":"formats/freeform.py","file_url":"https://github.com/ivnle/the-format-tax/blob/HEAD/formats/freeform.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"d093f593ac440765","mcp_get_code":{"code_sha256":"d093f593ac440765"}},{"arxiv_id":"2603.22582","paper":"/paper/arxiv-2603-22582","title":"LIE TO ME: HOW FAITHFUL IS CHAIN-OF-THOUGHT REASONING IN OPEN-WEIGHT REASONING MODELS?","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"ricyoung/cot-faithfulness-open-models","path":"src/cot_faithfulness/answer_extraction.py","file_url":"https://github.com/ricyoung/cot-faithfulness-open-models/blob/HEAD/src/cot_faithfulness/answer_extraction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"8735a844d15f3a2a","mcp_get_code":{"code_sha256":"8735a844d15f3a2a"}},{"arxiv_id":"2603.02146","paper":"/paper/arxiv-2603-02146","title":"LongRLVR: Long-Context Reinforcement Learning Requires Verifiable Context Rewards","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"real-absolute-AI/LongRLVR","path":"recipe/dapo/longrl_reward_manager.py","file_url":"https://github.com/real-absolute-AI/LongRLVR/blob/HEAD/recipe/dapo/longrl_reward_manager.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b9fa0381260e409d","mcp_get_code":{"code_sha256":"b9fa0381260e409d"}},{"arxiv_id":"2602.11509","paper":"/paper/arxiv-2602-11509","title":"Multimodal Fact-Level Attribution for Verifiable Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"meetdavidwan/murgat","path":"src/util.py","file_url":"https://github.com/meetdavidwan/murgat/blob/HEAD/src/util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ca419c78d00ec1d5","mcp_get_code":{"code_sha256":"ca419c78d00ec1d5"}},{"arxiv_id":"2602.07892","paper":"/paper/arxiv-2602-07892","title":"Safety Alignment as Continual Learning: Mitigating the Alignment Tax via Orthogonal Gradient Projection","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"SunGL001/OGPSA","path":"eval/GPQA_Diamond.py","file_url":"https://github.com/SunGL001/OGPSA/blob/HEAD/eval/GPQA_Diamond.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a7b81f2f5969d9ab","mcp_get_code":{"code_sha256":"a7b81f2f5969d9ab"}},{"arxiv_id":"2602.07892","paper":"/paper/arxiv-2602-07892","title":"Safety Alignment as Continual Learning: Mitigating the Alignment Tax via Orthogonal Gradient Projection","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"SunGL001/OGPSA","path":"eval/MMLU.py","file_url":"https://github.com/SunGL001/OGPSA/blob/HEAD/eval/MMLU.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b9ce5a37c260fee5","mcp_get_code":{"code_sha256":"b9ce5a37c260fee5"}},{"arxiv_id":"2602.03719","paper":"/paper/arxiv-2602-03719","title":"BranPO: Scalable Contrastive Branch Sampling for Long-Horizon Agentic Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"YubaoZhao/BranPO","path":"agent/search/llm_as_a_judge.py","file_url":"https://github.com/YubaoZhao/BranPO/blob/HEAD/agent/search/llm_as_a_judge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ec1077be0c2df0e6","mcp_get_code":{"code_sha256":"ec1077be0c2df0e6"}},{"arxiv_id":"2602.00428","paper":"/paper/arxiv-2602-00428","title":"When Agents \"Misremember\" Collectively: Exploring the Mandela Effect in LLM-based Multi-Agent Systems","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"bluedream02/Mandela-Effect","path":"eval_correct.py","file_url":"https://github.com/bluedream02/Mandela-Effect/blob/HEAD/eval_correct.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1f183be69c67626e","mcp_get_code":{"code_sha256":"1f183be69c67626e"}},{"arxiv_id":"2601.22139","paper":"/paper/arxiv-2601-22139","title":"Reasoning While Asking: Transforming Reasoning Large Language Models from Passive Solvers to Proactive Inquirers","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"SUAT-AIRI/Proactive-Interactive-R1","path":"generalization_eval/factual_knowledge/run_mmlu_interactive_generation.py","file_url":"https://github.com/SUAT-AIRI/Proactive-Interactive-R1/blob/HEAD/generalization_eval/factual_knowledge/run_mmlu_interactive_generation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"adf4a17f6030f2ef","mcp_get_code":{"code_sha256":"adf4a17f6030f2ef"}},{"arxiv_id":"2601.11866","paper":"/paper/arxiv-2601-11866","title":"Advances in LLM Reasoning Enable Flexibility in Clinical Problem-Solving","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"bernardolab/mARC-Reasoning","path":"compute_accuracy.py","file_url":"https://github.com/bernardolab/mARC-Reasoning/blob/HEAD/compute_accuracy.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1f59ab74392c3fcf","mcp_get_code":{"code_sha256":"1f59ab74392c3fcf"}},{"arxiv_id":"2511.18934","paper":"/paper/arxiv-2511-18934","title":"Skeletons Matter: Dynamic Data Augmentation for Text-to-Query","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"jjjycaptain/Skeletron","path":"src/data_syn.py","file_url":"https://github.com/jjjycaptain/Skeletron/blob/HEAD/src/data_syn.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"85b5611a1add5852","mcp_get_code":{"code_sha256":"85b5611a1add5852"}},{"arxiv_id":"2510.20342","paper":"/paper/arxiv-2510-20342","title":"Teaching Language Models to Reason with Tools","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"ChengpengLi1003/CoRT","path":"infer/parser.py","file_url":"https://github.com/ChengpengLi1003/CoRT/blob/HEAD/infer/parser.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"951c55f8667c735f","mcp_get_code":{"code_sha256":"951c55f8667c735f"}},{"arxiv_id":"2510.16416","paper":"/paper/arxiv-2510-16416","title":"SSL4RL: Revisiting Self-supervised Learning as Intrinsic Reward for Visual-Language Reasoning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"PKU-ML/SSL4RL","path":"verl/utils/reward_score/ssl4rl.py","file_url":"https://github.com/PKU-ML/SSL4RL/blob/HEAD/verl/utils/reward_score/ssl4rl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"93c07e2e0da6f774","mcp_get_code":{"code_sha256":"93c07e2e0da6f774"}},{"arxiv_id":"2510.11520","paper":"/paper/arxiv-2510-11520","title":"mmWalk: Towards Multi-modal Multi-view Walking Assistance","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"KediYing/mmWalk","path":"eval_gpt.py","file_url":"https://github.com/KediYing/mmWalk/blob/HEAD/eval_gpt.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"22f07d287593a1b8","mcp_get_code":{"code_sha256":"22f07d287593a1b8"}},{"arxiv_id":"2508.18847","paper":"/paper/arxiv-2508-18847","title":"ConfTuner: Training Large Language Models to Express Their Confidence Verbally","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"liushiliushi/ConfTuner","path":"src/llama_recipes/datasets2/gsm8k_dataset.py","file_url":"https://github.com/liushiliushi/ConfTuner/blob/HEAD/src/llama_recipes/datasets2/gsm8k_dataset.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d95f990735eba145","mcp_get_code":{"code_sha256":"d95f990735eba145"}},{"arxiv_id":"2507.02834","paper":null,"title":"arXiv:2507.02834","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"dhcode-cpp/X-R1","path":"src/x_r1/rewards.py","file_url":"https://github.com/dhcode-cpp/X-R1/blob/HEAD/src/x_r1/rewards.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ebe018f39b3bb7d5","mcp_get_code":{"code_sha256":"ebe018f39b3bb7d5"}},{"arxiv_id":"2507.02592","paper":"/paper/websailor-navigating-super-human-reasoning","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","date":"2025-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba-nlp/webwalker","path":"evaluation/evaluate_hle_official.py","file_url":"https://github.com/alibaba-nlp/webwalker/blob/HEAD/evaluation/evaluate_hle_official.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"079a31f347eec894","mcp_get_code":{"code_sha256":"079a31f347eec894"}},{"arxiv_id":"2506.20629","paper":"/paper/plop-precise-lora-placement-for-efficient","title":"PLoP: Precise LoRA Placement for Efficient Finetuning of Large Models","date":"2025-06-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"soufiane001/plop","path":"sft/eval_gsm8k.py","file_url":"https://github.com/soufiane001/plop/blob/HEAD/sft/eval_gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"54d8e73534679cb9","mcp_get_code":{"code_sha256":"54d8e73534679cb9"}},{"arxiv_id":"2506.18896","paper":"/paper/reasonflux-prm-trajectory-aware-prms-for-long","title":"ReasonFlux-PRM: Trajectory-Aware PRMs for Long Chain-of-Thought Reasoning in LLMs","date":"2025-06-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yangling0818/buffer-of-thought-llm","path":"meta_buffer_utilis.py","file_url":"https://github.com/yangling0818/buffer-of-thought-llm/blob/HEAD/meta_buffer_utilis.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"92d7b41138b8da5b","mcp_get_code":{"code_sha256":"92d7b41138b8da5b"}},{"arxiv_id":"2506.18105","paper":null,"title":"arXiv:2506.18105","date":null,"month_inferred_from_arxiv_id":"2025-06","title_source":null,"repo":"sofyc/ChengyuBench","path":"run-appropriateness.py","file_url":"https://github.com/sofyc/ChengyuBench/blob/HEAD/run-appropriateness.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c7bbfd05d008d535","mcp_get_code":{"code_sha256":"c7bbfd05d008d535"}},{"arxiv_id":"2506.18105","paper":null,"title":"arXiv:2506.18105","date":null,"month_inferred_from_arxiv_id":"2025-06","title_source":null,"repo":"sofyc/ChengyuBench","path":"run-connotation.py","file_url":"https://github.com/sofyc/ChengyuBench/blob/HEAD/run-connotation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e035e8b22820fb1a","mcp_get_code":{"code_sha256":"e035e8b22820fb1a"}},{"arxiv_id":"2506.14448","paper":"/paper/how-far-can-llms-improve-from-experience","title":"How Far Can LLMs Improve from Experience? Measuring Test-Time Learning Ability in LLMs with Human Comparison","date":"2025-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Alice1998/Test-time-Learning","path":"cumulative_setting/dynamic_cheatsheet/utils/extractor.py","file_url":"https://github.com/Alice1998/Test-time-Learning/blob/HEAD/cumulative_setting/dynamic_cheatsheet/utils/extractor.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef43d275e410158b","mcp_get_code":{"code_sha256":"ef43d275e410158b"}},{"arxiv_id":"2506.03230","paper":"/paper/diablo-diagonal-blocks-are-sufficient-for","title":"DiaBlo: Diagonal Blocks Are Sufficient For Finetuning","date":"2025-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ziyangjoy/diablo","path":"evaluate_commonsense.py","file_url":"https://github.com/ziyangjoy/diablo/blob/HEAD/evaluate_commonsense.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95f77b72bcc7ef90","mcp_get_code":{"code_sha256":"95f77b72bcc7ef90"}},{"arxiv_id":"2505.23754","paper":"/paper/deeptheorem-advancing-llm-reasoning-for","title":"DeepTheorem: Advancing LLM Reasoning for Theorem Proving Through Natural Language and Reinforcement Learning","date":"2025-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiahao004/deeptheorem","path":"eval/eval_llm.py","file_url":"https://github.com/jiahao004/deeptheorem/blob/HEAD/eval/eval_llm.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2b9cebd93558867a","mcp_get_code":{"code_sha256":"2b9cebd93558867a"}},{"arxiv_id":"2505.22961","paper":"/paper/tomap-training-opponent-aware-llm-persuaders","title":"ToMAP: Training Opponent-Aware LLM Persuaders with Theory of Mind","date":"2025-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ulab-uiuc/ToMAP","path":"verl/env_feedback/argument_graph.py","file_url":"https://github.com/ulab-uiuc/ToMAP/blob/HEAD/verl/env_feedback/argument_graph.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4a1bf7c49ac2d0c9","mcp_get_code":{"code_sha256":"4a1bf7c49ac2d0c9"}},{"arxiv_id":"2505.13308","paper":"/paper/seek-in-the-dark-reasoning-via-test-time","title":"Seek in the Dark: Reasoning via Test-Time Instance-Level Policy Gradient in Latent Space","date":"2025-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bigai-nlco/latentseek","path":"src/extract_judge_answer/utils.py","file_url":"https://github.com/bigai-nlco/latentseek/blob/HEAD/src/extract_judge_answer/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6b13e7afc2042384","mcp_get_code":{"code_sha256":"6b13e7afc2042384"}},{"arxiv_id":"2505.10518","paper":"/paper/multi-token-prediction-needs-registers","title":"Multi-Token Prediction Needs Registers","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nasosger/mutor","path":"language_modeling/src/eval/evaluate_gsm8k.py","file_url":"https://github.com/nasosger/mutor/blob/HEAD/language_modeling/src/eval/evaluate_gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0e4c30e9e6691a25","mcp_get_code":{"code_sha256":"0e4c30e9e6691a25"}},{"arxiv_id":"2505.07773","paper":"/paper/agent-rl-scaling-law-agent-rl-with","title":"Agent RL Scaling Law: Agent RL with Spontaneous Code Execution for Mathematical Problem Solving","date":"2025-05-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yyht/openrlhf_async_pipline","path":"evaluation/my_evaluation.py","file_url":"https://github.com/yyht/openrlhf_async_pipline/blob/HEAD/evaluation/my_evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4986aab3cd8d8212","mcp_get_code":{"code_sha256":"4986aab3cd8d8212"}},{"arxiv_id":"2505.00662","paper":"/paper/deepcritic-deliberate-critique-with-large","title":"DeepCritic: Deliberate Critique with Large Language Models","date":"2025-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucbm/deepcritic","path":"Critique_Generation/gen_step_solutions.py","file_url":"https://github.com/rucbm/deepcritic/blob/HEAD/Critique_Generation/gen_step_solutions.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"315054dfedd692ff","mcp_get_code":{"code_sha256":"315054dfedd692ff"}},{"arxiv_id":"2504.17454","paper":"/paper/adaptive-orchestration-of-modular-generative","title":"Adaptive Orchestration of Modular Generative Information Access Systems","date":"2025-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"informagi/AQA","path":"AQA_dataset_organizer.py","file_url":"https://github.com/informagi/AQA/blob/HEAD/AQA_dataset_organizer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"767285e542cae380","mcp_get_code":{"code_sha256":"767285e542cae380"}},{"arxiv_id":"2503.19707","paper":"/paper/mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stogiannidis/srbench","path":"src/eval/acc.py","file_url":"https://github.com/stogiannidis/srbench/blob/HEAD/src/eval/acc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"14a27694a27f4ad0","mcp_get_code":{"code_sha256":"14a27694a27f4ad0"}},{"arxiv_id":"2503.16188","paper":"/paper/cls-rl-image-classification-with-rule-based","title":"Think or Not Think: A Study of Explicit Thinking in Rule-Based Visual Reinforcement Fine-Tuning","date":"2025-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"minglllli/cls-rl","path":"src/eval/test_cvbench.py","file_url":"https://github.com/minglllli/cls-rl/blob/HEAD/src/eval/test_cvbench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2bb2ff42bd737b97","mcp_get_code":{"code_sha256":"2bb2ff42bd737b97"}},{"arxiv_id":"2503.07358","paper":"/paper/repost-scalable-repository-level-coding","title":"RepoST: Scalable Repository-Level Coding Environment Construction with Sandbox Testing","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiqingxyq/RepoST","path":"RepoST/llm_check.py","file_url":"https://github.com/yiqingxyq/RepoST/blob/HEAD/RepoST/llm_check.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"34dcaa09d95a53cd","mcp_get_code":{"code_sha256":"34dcaa09d95a53cd"}},{"arxiv_id":"2503.07003","paper":"/paper/large-language-models-often-say-one-thing-and","title":"Large Language Models Often Say One Thing and Do Another","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"icip-cas/Word-Deed-Consistency-Test","path":"eval/test_local_models.py","file_url":"https://github.com/icip-cas/Word-Deed-Consistency-Test/blob/HEAD/eval/test_local_models.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4ca8f0e08bcf3036","mcp_get_code":{"code_sha256":"4ca8f0e08bcf3036"}},{"arxiv_id":"2503.05188","paper":null,"title":"arXiv:2503.05188","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":"BugMakerzzz/CRISP","path":"crisp_reason.py","file_url":"https://github.com/BugMakerzzz/CRISP/blob/HEAD/crisp_reason.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"82ee443d76f92d28","mcp_get_code":{"code_sha256":"82ee443d76f92d28"}},{"arxiv_id":"2503.05132","paper":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","date":"2025-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2bb2ff42bd737b97","mcp_get_code":{"code_sha256":"2bb2ff42bd737b97"}},{"arxiv_id":"2502.19187","paper":"/paper/big-bench-extra-hard","title":"BIG-Bench Extra Hard","date":"2025-02-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-deepmind/bbeh","path":"bbeh/evaluate.py","file_url":"https://github.com/google-deepmind/bbeh/blob/HEAD/bbeh/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1148b66bf0848e3d","mcp_get_code":{"code_sha256":"1148b66bf0848e3d"}},{"arxiv_id":"2502.06781","paper":"/paper/exploring-the-limit-of-outcome-reward-for","title":"Exploring the Limit of Outcome Reward for Learning Mathematical Reasoning","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"internlm/oreal","path":"oreal/judgers/utils.py","file_url":"https://github.com/internlm/oreal/blob/HEAD/oreal/judgers/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0e8f4b09eb5b8c47","mcp_get_code":{"code_sha256":"0e8f4b09eb5b8c47"}},{"arxiv_id":"2502.06703","paper":"/paper/can-1b-llm-surpass-405b-llm-rethinking","title":"Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RyanLiu112/compute-optimal-tts","path":"src/envs/MATH/env.py","file_url":"https://github.com/RyanLiu112/compute-optimal-tts/blob/HEAD/src/envs/MATH/env.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"042094830d54a879","mcp_get_code":{"code_sha256":"042094830d54a879"}},{"arxiv_id":"2502.02384","paper":"/paper/stair-improving-safety-alignment-with","title":"STAIR: Improving Safety Alignment with Introspective Reasoning","date":"2025-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-ml/stair","path":"src/final_orm.py","file_url":"https://github.com/thu-ml/stair/blob/HEAD/src/final_orm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fcfc0ecd7309f71a","mcp_get_code":{"code_sha256":"fcfc0ecd7309f71a"}},{"arxiv_id":"2501.13381","paper":"/paper/do-as-we-do-not-as-you-think-the-conformity","title":"Do as We Do, Not as You Think: the Conformity of Large Language Models","date":"2025-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Zhiyuan-Weng/BenchForm","path":"reflection.py","file_url":"https://github.com/Zhiyuan-Weng/BenchForm/blob/HEAD/reflection.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1f183be69c67626e","mcp_get_code":{"code_sha256":"1f183be69c67626e"}},{"arxiv_id":"2501.04962","paper":"/paper/voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dreamtheater123/voxeval","path":"metric_calculation.py","file_url":"https://github.com/dreamtheater123/voxeval/blob/HEAD/metric_calculation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95d8414ac0a018a0","mcp_get_code":{"code_sha256":"95d8414ac0a018a0"}},{"arxiv_id":"2412.15204","paper":"/paper/longbench-v2-towards-deeper-understanding-and","title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","date":"2024-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2807cfec4fc4d41e","mcp_get_code":{"code_sha256":"2807cfec4fc4d41e"}},{"arxiv_id":"2412.13871","paper":"/paper/llava-uhd-v2-an-mllm-integrating-high","title":"LLaVA-UHD v2: an MLLM Integrating High-Resolution Feature Pyramid via Hierarchical Window Transformer","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/llava-uhd","path":"VLMEvalKit/vlmeval/vlm/llava_qwen2_uhd_v3.py","file_url":"https://github.com/thunlp/llava-uhd/blob/HEAD/VLMEvalKit/vlmeval/vlm/llava_qwen2_uhd_v3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a76fe5da7ade58fb","mcp_get_code":{"code_sha256":"a76fe5da7ade58fb"}},{"arxiv_id":"2412.08819","paper":"/paper/harp-a-challenging-human-annotated-math","title":"HARP: A challenging human-annotated math reasoning benchmark","date":"2024-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aadityasingh/harp","path":"src/eval/parsing_lib.py","file_url":"https://github.com/aadityasingh/harp/blob/HEAD/src/eval/parsing_lib.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cf5843ce85c71a69","mcp_get_code":{"code_sha256":"cf5843ce85c71a69"}},{"arxiv_id":"2412.06559","paper":"/paper/processbench-identifying-process-errors-in","title":"ProcessBench: Identifying Process Errors in Mathematical Reasoning","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qwenlm/processbench","path":"code/run_eval.py","file_url":"https://github.com/qwenlm/processbench/blob/HEAD/code/run_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"315054dfedd692ff","mcp_get_code":{"code_sha256":"315054dfedd692ff"}},{"arxiv_id":"2410.17885","paper":"/paper/r-cot-reverse-chain-of-thought-problem","title":"R-CoT: Reverse Chain-of-Thought Problem Generation for Geometric Reasoning in Large Multimodal Models","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dle666/r-cot","path":"MathVista_eval/evaluation/extract_answer.py","file_url":"https://github.com/dle666/r-cot/blob/HEAD/MathVista_eval/evaluation/extract_answer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c1eaa5bf4d5f383d","mcp_get_code":{"code_sha256":"c1eaa5bf4d5f383d"}},{"arxiv_id":"2410.15700","paper":"/paper/internlm2-5-stepprover-advancing-automated","title":"InternLM2.5-StepProver: Advancing Automated Theorem Proving via Expert Iteration on Large-Scale LEAN Problems","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"internlm/internlm-math","path":"agent/math_agent.py","file_url":"https://github.com/internlm/internlm-math/blob/HEAD/agent/math_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6cd811d1cce3b5e7","mcp_get_code":{"code_sha256":"6cd811d1cce3b5e7"}},{"arxiv_id":"2410.04838","paper":"/paper/rationale-aware-answer-verification-by","title":"Rationale-Aware Answer Verification by Pairwise Self-Evaluation","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"akirakawabata/reps","path":"src/reward_training.py","file_url":"https://github.com/akirakawabata/reps/blob/HEAD/src/reward_training.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"747f8df0daf7cfe2","mcp_get_code":{"code_sha256":"747f8df0daf7cfe2"}},{"arxiv_id":"2409.00055","paper":"/paper/sorsa-singular-values-and-orthonormal","title":"SORSA: Singular Values and Orthonormal Regularized Singular Vectors Adaptation of Large Language Models","date":"2024-08-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Gunale0926/SORSA","path":"dataset.py","file_url":"https://github.com/Gunale0926/SORSA/blob/HEAD/dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c923d19564d2546f","mcp_get_code":{"code_sha256":"c923d19564d2546f"}},{"arxiv_id":"2407.21783","paper":"/paper/the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wenet-e2e/west","path":"west/bin/decode_mmau.py","file_url":"https://github.com/wenet-e2e/west/blob/HEAD/west/bin/decode_mmau.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0a0d3ec6fb8f3c46","mcp_get_code":{"code_sha256":"0a0d3ec6fb8f3c46"}},{"arxiv_id":"2407.21783","paper":"/paper/the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wenet-e2e/west","path":"west/bin/decode_mmsu.py","file_url":"https://github.com/wenet-e2e/west/blob/HEAD/west/bin/decode_mmsu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d7b280c2f396affc","mcp_get_code":{"code_sha256":"d7b280c2f396affc"}},{"arxiv_id":"2407.16364","paper":"/paper/harmonizing-visual-text-comprehension-and","title":"Harmonizing Visual Text Comprehension and Generation","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/textharmony","path":"TextHarmony/utils/vqa_score.py","file_url":"https://github.com/bytedance/textharmony/blob/HEAD/TextHarmony/utils/vqa_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c1b06375bdb265f4","mcp_get_code":{"code_sha256":"c1b06375bdb265f4"}},{"arxiv_id":"2407.12854","paper":"/paper/scaling-retrieval-based-language-models-with","title":"Scaling Retrieval-Based Language Models with a Trillion-Token Datastore","date":"2024-07-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rulinshao/retrieval-scaling","path":"src/evaluate_perplexity.py","file_url":"https://github.com/rulinshao/retrieval-scaling/blob/HEAD/src/evaluate_perplexity.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"290c5134d15244b3","mcp_get_code":{"code_sha256":"290c5134d15244b3"}},{"arxiv_id":"2407.12784","paper":"/paper/agentpoison-red-teaming-llm-agents-via","title":"AgentPoison: Red-teaming LLM Agents via Poisoning Memory or Knowledge Bases","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BillChan226/AgentPoison","path":"ReAct/search.py","file_url":"https://github.com/BillChan226/AgentPoison/blob/HEAD/ReAct/search.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fcc271edc48aad1c","mcp_get_code":{"code_sha256":"fcc271edc48aad1c"}},{"arxiv_id":"2406.18403","paper":"/paper/llms-instead-of-human-judges-a-large-scale","title":"LLMs instead of Human Judges? A Large Scale Empirical Study across 20 NLP Evaluation Tasks","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmg-illc/JUDGE-BENCH","path":"eval_responses.py","file_url":"https://github.com/dmg-illc/JUDGE-BENCH/blob/HEAD/eval_responses.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4203c0dc81afe24e","mcp_get_code":{"code_sha256":"4203c0dc81afe24e"}},{"arxiv_id":"2406.16797","paper":"/paper/lottery-ticket-adaptation-mitigating","title":"Lottery Ticket Adaptation: Mitigating Destructive Interference in LLMs","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kiddyboots216/lottery-ticket-adaptation","path":"rlaif/eval_model_all.py","file_url":"https://github.com/kiddyboots216/lottery-ticket-adaptation/blob/HEAD/rlaif/eval_model_all.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bd0cc71c62b4a212","mcp_get_code":{"code_sha256":"bd0cc71c62b4a212"}},{"arxiv_id":"2406.12809","paper":"/paper/can-large-language-models-always-solve-easy","title":"Can Large Language Models Always Solve Easy Problems if They Can Solve Harder Ones?","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QwenLM/ConsisEval","path":"math_check/check.py","file_url":"https://github.com/QwenLM/ConsisEval/blob/HEAD/math_check/check.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b22b268ce5bb5b68","mcp_get_code":{"code_sha256":"b22b268ce5bb5b68"}},{"arxiv_id":"2406.07791","paper":"/paper/judging-the-judges-a-systematic-investigation","title":"Judging the Judges: A Systematic Study of Position Bias in LLM-as-a-Judge","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Slimshilin/Position-Bias-Analyzer-Demo","path":"position_bias_analyzer/calculate_consistency_and_preference/consistency_preference_util.py","file_url":"https://github.com/Slimshilin/Position-Bias-Analyzer-Demo/blob/HEAD/position_bias_analyzer/calculate_consistency_and_preference/consistency_preference_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"726da9cc7c6c575a","mcp_get_code":{"code_sha256":"726da9cc7c6c575a"}},{"arxiv_id":"2406.06469","paper":"/paper/husky-a-unified-open-source-language-agent","title":"Husky: A Unified, Open-Source Language Agent for Multi-Step Reasoning","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"agent-husky/husky-v1","path":"husky/run_husky.py","file_url":"https://github.com/agent-husky/husky-v1/blob/HEAD/husky/run_husky.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bf89f824c7dab150","mcp_get_code":{"code_sha256":"bf89f824c7dab150"}},{"arxiv_id":"2406.04271","paper":"/paper/buffer-of-thoughts-thought-augmented","title":"Buffer of Thoughts: Thought-Augmented Reasoning with Large Language Models","date":"2024-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YangLing0818/buffer-of-thought-llm","path":"meta_buffer_utilis.py","file_url":"https://github.com/YangLing0818/buffer-of-thought-llm/blob/HEAD/meta_buffer_utilis.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"92d7b41138b8da5b","mcp_get_code":{"code_sha256":"92d7b41138b8da5b"}},{"arxiv_id":"2406.01574","paper":"/paper/mmlu-pro-a-more-robust-and-challenging-multi","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","date":"2024-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tiger-ai-lab/mmlu-pro","path":"compute_accuracy.py","file_url":"https://github.com/tiger-ai-lab/mmlu-pro/blob/HEAD/compute_accuracy.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1f59ab74392c3fcf","mcp_get_code":{"code_sha256":"1f59ab74392c3fcf"}},{"arxiv_id":"2405.20978","paper":"/paper/enhancing-noise-robustness-of-retrieval","title":"Enhancing Noise Robustness of Retrieval-Augmented Language Models with Adaptive Adversarial Training","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"calubkk/RAAT","path":"tuner/utils/answer_processor.py","file_url":"https://github.com/calubkk/RAAT/blob/HEAD/tuner/utils/answer_processor.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fa90936b1f466162","mcp_get_code":{"code_sha256":"fa90936b1f466162"}},{"arxiv_id":"2405.15232","paper":"/paper/deem-diffusion-models-serve-as-the-eyes-of","title":"DEEM: Diffusion Models Serve as the Eyes of Large Language Models for Image Perception","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rainbowluocs/deem","path":"uni_interleaved/utils/vqa_score.py","file_url":"https://github.com/rainbowluocs/deem/blob/HEAD/uni_interleaved/utils/vqa_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c1b06375bdb265f4","mcp_get_code":{"code_sha256":"c1b06375bdb265f4"}},{"arxiv_id":"2405.14838","paper":"/paper/from-explicit-cot-to-implicit-cot-learning-to","title":"From Explicit CoT to Implicit CoT: Learning to Internalize CoT Step by Step","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"da03/internalize_cot_step_by_step","path":"src/data.py","file_url":"https://github.com/da03/internalize_cot_step_by_step/blob/HEAD/src/data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4d589052839e4a5e","mcp_get_code":{"code_sha256":"4d589052839e4a5e"}},{"arxiv_id":"2405.00722","paper":"/paper/llms-for-generating-and-evaluating","title":"LLMs for Generating and Evaluating Counterfactuals: A Comprehensive Study","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aix-group/llms-for-cfs","path":"src/gen_cf/generate_hatespeech.py","file_url":"https://github.com/aix-group/llms-for-cfs/blob/HEAD/src/gen_cf/generate_hatespeech.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ed21c4ca2697e0d9","mcp_get_code":{"code_sha256":"ed21c4ca2697e0d9"}},{"arxiv_id":"2405.00722","paper":"/paper/llms-for-generating-and-evaluating","title":"LLMs for Generating and Evaluating Counterfactuals: A Comprehensive Study","date":"2024-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aix-group/llms-for-cfs","path":"src/gen_cf/generate_imdb.py","file_url":"https://github.com/aix-group/llms-for-cfs/blob/HEAD/src/gen_cf/generate_imdb.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"be81c1789179c1ac","mcp_get_code":{"code_sha256":"be81c1789179c1ac"}},{"arxiv_id":"2404.10346","paper":"/paper/self-explore-to-avoid-the-pit-improving-the","title":"Self-Explore: Enhancing Mathematical Reasoning in Language Models with Fine-grained Rewards","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hbin0701/Self-Explore","path":"gen/utils_others.py","file_url":"https://github.com/hbin0701/Self-Explore/blob/HEAD/gen/utils_others.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"50408bcb3e87455b","mcp_get_code":{"code_sha256":"50408bcb3e87455b"}},{"arxiv_id":"2404.08700","paper":"/paper/is-your-llm-outdated-benchmarking-llms","title":"DyKnow: Dynamically Verifying Time-Sensitive Factual Knowledge in LLMs","date":"2024-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sislab-unitn/dyknow","path":"models_output/analyze_replies.py","file_url":"https://github.com/sislab-unitn/dyknow/blob/HEAD/models_output/analyze_replies.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"17600052c8ca574e","mcp_get_code":{"code_sha256":"17600052c8ca574e"}},{"arxiv_id":"2403.17359","paper":"/paper/chain-of-action-faithful-and-multimodal","title":"Chain-of-Action: Faithful and Multimodal Question Answering through Large Language Models","date":"2024-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MAGICS-LAB/Chain-of-Actions","path":"chain-of-search-wo-ir.py","file_url":"https://github.com/MAGICS-LAB/Chain-of-Actions/blob/HEAD/chain-of-search-wo-ir.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"047a7d86ce36d383","mcp_get_code":{"code_sha256":"047a7d86ce36d383"}},{"arxiv_id":"2403.05307","paper":"/paper/tapilot-crossing-benchmarking-and-evolving","title":"Tapilot-Crossing: Benchmarking and Evolving LLMs Towards Interactive Data Analysis Agents","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tapilot-crossing/tapilot_code","path":"eval/eval_multi_choice.py","file_url":"https://github.com/tapilot-crossing/tapilot_code/blob/HEAD/eval/eval_multi_choice.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8c5a9f85651f57df","mcp_get_code":{"code_sha256":"8c5a9f85651f57df"}},{"arxiv_id":"2403.04706","paper":"/paper/common-7b-language-models-already-possess","title":"Common 7B Language Models Already Possess Strong Math Capabilities","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jerrywu-code/susgen","path":"eval/code/eval_finqa.py","file_url":"https://github.com/jerrywu-code/susgen/blob/HEAD/eval/code/eval_finqa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"30e951ff25cd61d7","mcp_get_code":{"code_sha256":"30e951ff25cd61d7"}},{"arxiv_id":"2402.17709","paper":"/paper/case-based-or-rule-based-how-do-transformers","title":"Case-Based or Rule-Based: How Do Transformers Do the Math?","date":"2024-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"graphpku/case_or_rule","path":"datasets/addition/cot_gen.py","file_url":"https://github.com/graphpku/case_or_rule/blob/HEAD/datasets/addition/cot_gen.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fd391d2d2217abf1","mcp_get_code":{"code_sha256":"fd391d2d2217abf1"}},{"arxiv_id":"2402.15052","paper":"/paper/tombench-benchmarking-theory-of-mind-in-large","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","date":"2024-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhchen18/tombench","path":"get_results.py","file_url":"https://github.com/zhchen18/tombench/blob/HEAD/get_results.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"048418fb06a97048","mcp_get_code":{"code_sha256":"048418fb06a97048"}},{"arxiv_id":"2402.15000","paper":"/paper/divide-or-conquer-which-part-should-you","title":"Divide-or-Conquer? Which Part Should You Distill Your LLM?","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apple/ml-divide-or-conquer","path":"evaluation/evaluation.py","file_url":"https://github.com/apple/ml-divide-or-conquer/blob/HEAD/evaluation/evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"2c7f4d404b0a3c5b","mcp_get_code":{"code_sha256":"2c7f4d404b0a3c5b"}},{"arxiv_id":"2402.15000","paper":"/paper/divide-or-conquer-which-part-should-you","title":"Divide-or-Conquer? Which Part Should You Distill Your LLM?","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apple/ml-divide-or-conquer","path":"evaluation/evaluation_bamb.py","file_url":"https://github.com/apple/ml-divide-or-conquer/blob/HEAD/evaluation/evaluation_bamb.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"9efb8ade48dbb583","mcp_get_code":{"code_sha256":"9efb8ade48dbb583"}},{"arxiv_id":"2402.14008","paper":"/paper/olympiadbench-a-challenging-benchmark-for","title":"OlympiadBench: A Challenging Benchmark for Promoting AGI with Olympiad-Level Bilingual Multimodal Scientific Problems","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OpenBMB/OlympiadBench","path":"inference/judge.py","file_url":"https://github.com/OpenBMB/OlympiadBench/blob/HEAD/inference/judge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5328984c0ec57f93","mcp_get_code":{"code_sha256":"5328984c0ec57f93"}},{"arxiv_id":"2402.08219","paper":"/paper/bbox-adapter-lightweight-adapting-for-black","title":"BBox-Adapter: Lightweight Adapting for Black-Box Large Language Models","date":"2024-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haotiansun14/bbox-adapter","path":"utils/gsm8k_metric.py","file_url":"https://github.com/haotiansun14/bbox-adapter/blob/HEAD/utils/gsm8k_metric.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6731183eb0c6819d","mcp_get_code":{"code_sha256":"6731183eb0c6819d"}},{"arxiv_id":"2402.06925","paper":"/paper/a-thorough-examination-of-decoding-methods-in","title":"A Thorough Examination of Decoding Methods in the Era of LLMs","date":"2024-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"davidfanzz/llm_decoding","path":"lm_eval/tasks/gsm8k.py","file_url":"https://github.com/davidfanzz/llm_decoding/blob/HEAD/lm_eval/tasks/gsm8k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ac208e27f559cfcb","mcp_get_code":{"code_sha256":"ac208e27f559cfcb"}},{"arxiv_id":"2401.12954","paper":"/paper/meta-prompting-enhancing-language-models-with","title":"Meta-Prompting: Enhancing Language Models with Task-Agnostic Scaffolding","date":"2024-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"suzgunmirac/meta-prompting","path":"evaluate_outputs.py","file_url":"https://github.com/suzgunmirac/meta-prompting/blob/HEAD/evaluate_outputs.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"58a6661987e8b91b","mcp_get_code":{"code_sha256":"58a6661987e8b91b"}},{"arxiv_id":"2401.11458","paper":"/paper/linear-alignment-a-closed-form-solution-for","title":"Linear Alignment: A Closed-form Solution for Aligning Human Preferences without Tuning and Feedback","date":"2024-01-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wizardcoast/linear_alignment","path":"preference_eval.py","file_url":"https://github.com/wizardcoast/linear_alignment/blob/HEAD/preference_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b4b74510d16a049c","mcp_get_code":{"code_sha256":"b4b74510d16a049c"}},{"arxiv_id":"2401.10480","paper":"/paper/escape-sky-high-cost-early-stopping-self","title":"Escape Sky-high Cost: Early-stopping Self-Consistency for Multi-step Reasoning","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yiwei98/ESC","path":"consistency_gsm8k.py","file_url":"https://github.com/Yiwei98/ESC/blob/HEAD/consistency_gsm8k.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d95f990735eba145","mcp_get_code":{"code_sha256":"d95f990735eba145"}},{"arxiv_id":"2401.10480","paper":"/paper/escape-sky-high-cost-early-stopping-self","title":"Escape Sky-high Cost: Early-stopping Self-Consistency for Multi-step Reasoning","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yiwei98/ESC","path":"consistency_coin.py","file_url":"https://github.com/Yiwei98/ESC/blob/HEAD/consistency_coin.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6ec18dd15818d034","mcp_get_code":{"code_sha256":"6ec18dd15818d034"}},{"arxiv_id":"2401.10480","paper":"/paper/escape-sky-high-cost-early-stopping-self","title":"Escape Sky-high Cost: Early-stopping Self-Consistency for Multi-step Reasoning","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yiwei98/ESC","path":"consistency_csqa.py","file_url":"https://github.com/Yiwei98/ESC/blob/HEAD/consistency_csqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b550af4a369be974","mcp_get_code":{"code_sha256":"b550af4a369be974"}},{"arxiv_id":"2401.10480","paper":"/paper/escape-sky-high-cost-early-stopping-self","title":"Escape Sky-high Cost: Early-stopping Self-Consistency for Multi-step Reasoning","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yiwei98/ESC","path":"consistency_last.py","file_url":"https://github.com/Yiwei98/ESC/blob/HEAD/consistency_last.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"691cea85ad41fe0e","mcp_get_code":{"code_sha256":"691cea85ad41fe0e"}},{"arxiv_id":"2312.02051","paper":"/paper/timechat-a-time-sensitive-multimodal-large","title":"TimeChat: A Time-sensitive Multimodal Large Language Model for Long Video Understanding","date":"2023-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lntzm/cvpr24track-longvideo","path":"benchmark/evaluate_egoschema.py","file_url":"https://github.com/lntzm/cvpr24track-longvideo/blob/HEAD/benchmark/evaluate_egoschema.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"8c3317746d989d00","mcp_get_code":{"code_sha256":"8c3317746d989d00"}},{"arxiv_id":"2311.08711","paper":"/paper/plug-leveraging-pivot-language-in-cross","title":"PLUG: Leveraging Pivot Language in Cross-Lingual Instruction Tuning","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ytyz1307zzh/plug","path":"src/evaluate/svamp/evaluate_svamp.py","file_url":"https://github.com/ytyz1307zzh/plug/blob/HEAD/src/evaluate/svamp/evaluate_svamp.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3e78d709f22fd30c","mcp_get_code":{"code_sha256":"3e78d709f22fd30c"}},{"arxiv_id":"2311.01460","paper":"/paper/implicit-chain-of-thought-reasoning-via","title":"Implicit Chain of Thought Reasoning via Knowledge Distillation","date":"2023-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"da03/implicit_chain_of_thought","path":"src/data.py","file_url":"https://github.com/da03/implicit_chain_of_thought/blob/HEAD/src/data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4d589052839e4a5e","mcp_get_code":{"code_sha256":"4d589052839e4a5e"}},{"arxiv_id":"2310.07177","paper":"/paper/online-speculative-decoding","title":"Online Speculative Decoding","date":"2023-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuxiaoxuanpku/osd","path":"distill/data.py","file_url":"https://github.com/liuxiaoxuanpku/osd/blob/HEAD/distill/data.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d95f990735eba145","mcp_get_code":{"code_sha256":"d95f990735eba145"}},{"arxiv_id":"2310.06147","paper":"/paper/reinforcement-learning-in-the-era-of-llms","title":"Reinforcement Learning in the Era of LLMs: What is Essential? What is needed? An RL Perspective on RLHF, Prompting, and Beyond","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vanderschaarlab/prompt-oirl","path":"llama_exps/llama_step1_gen_offline.py","file_url":"https://github.com/vanderschaarlab/prompt-oirl/blob/HEAD/llama_exps/llama_step1_gen_offline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d95f990735eba145","mcp_get_code":{"code_sha256":"d95f990735eba145"}},{"arxiv_id":"2310.02107","paper":"/paper/instance-needs-more-care-rewriting-prompts","title":"Instances Need More Care: Rewriting Prompts for Instances with LLMs in the Loop Yields Better Zero-Shot Performance","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salokr/propmted","path":"PRomPTed.py","file_url":"https://github.com/salokr/propmted/blob/HEAD/PRomPTed.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"625b937bea14125f","mcp_get_code":{"code_sha256":"625b937bea14125f"}},{"arxiv_id":"2309.17452","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","date":"2023-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/tora","path":"src/utils/parser.py","file_url":"https://github.com/microsoft/tora/blob/HEAD/src/utils/parser.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"951c55f8667c735f","mcp_get_code":{"code_sha256":"951c55f8667c735f"}},{"arxiv_id":"2309.06553","paper":"/paper/offline-prompt-evaluation-and-optimization","title":"Query-Dependent Prompt Evaluation and Optimization with Offline Inverse RL","date":"2023-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"d95f990735eba145","mcp_get_code":{"code_sha256":"d95f990735eba145"}},{"arxiv_id":"2308.16463","paper":"/paper/sparkles-unlocking-chats-across-multiple","title":"Sparkles: Unlocking Chats Across Multiple Images for Multimodal Instruction-Following Models","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hypjudy/sparkles","path":"evaluate.py","file_url":"https://github.com/hypjudy/sparkles/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ef4e70f7a2ab108","mcp_get_code":{"code_sha256":"0ef4e70f7a2ab108"}},{"arxiv_id":"2308.14508","paper":"/paper/longbench-a-bilingual-multitask-benchmark-for","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","date":"2023-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/longbench","path":"pred.py","file_url":"https://github.com/thudm/longbench/blob/HEAD/pred.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2807cfec4fc4d41e","mcp_get_code":{"code_sha256":"2807cfec4fc4d41e"}},{"arxiv_id":"2308.01825","paper":"/paper/scaling-relationship-on-learning-mathematical","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","date":"2023-08-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ofa-sys/gsm8k-screl","path":"collect_rejection_sampling.py","file_url":"https://github.com/ofa-sys/gsm8k-screl/blob/HEAD/collect_rejection_sampling.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e94e204600c3de88","mcp_get_code":{"code_sha256":"e94e204600c3de88"}},{"arxiv_id":"2307.07697","paper":"/paper/think-on-graph-deep-and-responsible-reasoning","title":"Think-on-Graph: Deep and Responsible Reasoning of Large Language Model on Knowledge Graph","date":"2023-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gasolsun36/tog","path":"ToG/wiki_func.py","file_url":"https://github.com/gasolsun36/tog/blob/HEAD/ToG/wiki_func.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"93ab9efdb03f32cb","mcp_get_code":{"code_sha256":"93ab9efdb03f32cb"}},{"arxiv_id":"2305.04388","paper":"/paper/language-models-don-t-always-say-what-they-1","title":"Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting","date":"2023-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"milesaturpin/cot-unfaithfulness","path":"run_eval.py","file_url":"https://github.com/milesaturpin/cot-unfaithfulness/blob/HEAD/run_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"083e331d55fb470d","mcp_get_code":{"code_sha256":"083e331d55fb470d"}},{"arxiv_id":"2305.04388","paper":"/paper/language-models-don-t-always-say-what-they-1","title":"Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting","date":"2023-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"milesaturpin/cot-unfaithfulness","path":"bbq_analysis.py","file_url":"https://github.com/milesaturpin/cot-unfaithfulness/blob/HEAD/bbq_analysis.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2ff6877b40dfd2bf","mcp_get_code":{"code_sha256":"2ff6877b40dfd2bf"}},{"arxiv_id":"openreview_QrC8OgQyOI","paper":null,"title":"arXiv:openreview_QrC8OgQyOI","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"viiika/Prism","path":"Dream/Dream_Prism/metrics/math500_eval.py","file_url":"https://github.com/viiika/Prism/blob/HEAD/Dream/Dream_Prism/metrics/math500_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5e56a4223b47eebb","mcp_get_code":{"code_sha256":"5e56a4223b47eebb"}},{"arxiv_id":"Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","paper":null,"title":"arXiv:Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"guozix/DVLR","path":"eval/MathVista/evaluation/extract_answer.py","file_url":"https://github.com/guozix/DVLR/blob/HEAD/eval/MathVista/evaluation/extract_answer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c1eaa5bf4d5f383d","mcp_get_code":{"code_sha256":"c1eaa5bf4d5f383d"}},{"arxiv_id":"Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","paper":null,"title":"arXiv:Guo_Integrating_Visual_Interpretation_and_Linguistic_Reasoning_for_Geometric_Problem_Solving_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"guozix/DVLR","path":"eval/MathVerse/evaluation/extract_answer.py","file_url":"https://github.com/guozix/DVLR/blob/HEAD/eval/MathVerse/evaluation/extract_answer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"659edf1412988476","mcp_get_code":{"code_sha256":"659edf1412988476"}},{"arxiv_id":"2025.findings-acl.301","paper":null,"title":"arXiv:2025.findings-acl.301","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenBMB/ConsJudge","path":"src/ConsJudge_train/embedding_similarity.py","file_url":"https://github.com/OpenBMB/ConsJudge/blob/HEAD/src/ConsJudge_train/embedding_similarity.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0907fc5b78db3678","mcp_get_code":{"code_sha256":"0907fc5b78db3678"}}]}