{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/parse-answer","entry":"parse_answer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":29,"n_papers_ran":16,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":29,"n_samples_ran":14,"n_samples_fingerprinted":6,"n_places":31,"n_places_pointer_only":13,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":10,"unverified":15},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.31108","paper":"/paper/arxiv-2608-31108","title":"Stress-Testing Efficient Responsible-AI Evaluation: When Compute Savings Change Benchmark Conclusions","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"VectorInstitute/sustainable-rai-evaluation","path":"src/evaluation_has_a_footprint/parsing.py","file_url":"https://github.com/VectorInstitute/sustainable-rai-evaluation/blob/HEAD/src/evaluation_has_a_footprint/parsing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"85a97cdb29eadeb1","mcp_get_code":{"code_sha256":"85a97cdb29eadeb1"}},{"arxiv_id":"2608.25655","paper":"/paper/arxiv-2608-25655","title":"Reconstructing the Right Episode: Evaluating Interleaved Conversational Memory Beyond Long Context","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"LordTARN1SHED/SCALE-QA","path":"tsim_reference/src/eval/metrics.py","file_url":"https://github.com/LordTARN1SHED/SCALE-QA/blob/HEAD/tsim_reference/src/eval/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"e22500cf37ccce0d","mcp_get_code":{"code_sha256":"e22500cf37ccce0d"}},{"arxiv_id":"2607.26873","paper":"/paper/arxiv-2607-26873","title":"SERPO: Self-Evolving Rubric Policy Optimization for Open-Ended Test-Time Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"chiefovoavicii/SERPO","path":"eval/eval_gpqa_diamond.py","file_url":"https://github.com/chiefovoavicii/SERPO/blob/HEAD/eval/eval_gpqa_diamond.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a1fadfeff49766bf","mcp_get_code":{"code_sha256":"a1fadfeff49766bf"}},{"arxiv_id":"2604.19459","paper":"/paper/arxiv-2604-19459","title":"Do LLMs Game Formalization? Evaluating Faithfulness in Logical Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"koreankiwi99/formalization-gaming","path":"src/utils/answer_parsing.py","file_url":"https://github.com/koreankiwi99/formalization-gaming/blob/HEAD/src/utils/answer_parsing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5200c424eaeecdfc","mcp_get_code":{"code_sha256":"5200c424eaeecdfc"}},{"arxiv_id":"2602.03645","paper":"/paper/arxiv-2602-03645","title":"Reinforcement Fine-Tuning for History-Aware Dense Retriever in RAG","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"zyc140345/HARR","path":"llama_index_hacked/query_engine.py","file_url":"https://github.com/zyc140345/HARR/blob/HEAD/llama_index_hacked/query_engine.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e406373defafb0db","mcp_get_code":{"code_sha256":"e406373defafb0db"}},{"arxiv_id":"2602.01030","paper":"/paper/arxiv-2602-01030","title":"Bias in the Ear of the Listener: Assessing Sensitivity in Audio Language Models Across Linguistic, Demographic, and Positional Variations","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ntunlplab/BiasInEar","path":"src/biasinear/models/_parser.py","file_url":"https://github.com/ntunlplab/BiasInEar/blob/HEAD/src/biasinear/models/_parser.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2f2d25b6d159ef94","mcp_get_code":{"code_sha256":"2f2d25b6d159ef94"}},{"arxiv_id":"2601.13304","paper":"/paper/arxiv-2601-13304","title":"CausalSpatial: A Benchmark for Object-Centric Causal Spatial Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"CausalSpatial/CausalSpatial","path":"score.py","file_url":"https://github.com/CausalSpatial/CausalSpatial/blob/HEAD/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cdaae78194e1ac94","mcp_get_code":{"code_sha256":"cdaae78194e1ac94"}},{"arxiv_id":"2601.05075","paper":"/paper/arxiv-2601-05075","title":"SemPA: Improving Sentence Embeddings of Large Language Models through Semantic Preference Alignment","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"szu-tera/SemPA","path":"drop/utils.py","file_url":"https://github.com/szu-tera/SemPA/blob/HEAD/drop/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"082dc362726bba49","mcp_get_code":{"code_sha256":"082dc362726bba49"}},{"arxiv_id":"2510.06307","paper":"/paper/arxiv-2510-06307","title":"Belief-Calibrated Multi-Agent Consensus Seeking for Complex NLP Tasks","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"dengwentao99/BCCS","path":"MMLU/evaluate.py","file_url":"https://github.com/dengwentao99/BCCS/blob/HEAD/MMLU/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7548ef78c115707c","mcp_get_code":{"code_sha256":"7548ef78c115707c"}},{"arxiv_id":"2510.03215","paper":"/paper/arxiv-2510-03215","title":"Cache-to-Cache: Direct Semantic Communication Between Large Language Models","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"thu-nics/C2C","path":"rosetta/utils/evaluate.py","file_url":"https://github.com/thu-nics/C2C/blob/HEAD/rosetta/utils/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6a9ab19801c22746","mcp_get_code":{"code_sha256":"6a9ab19801c22746"}},{"arxiv_id":"2505.10981","paper":"/paper/2505-10981","title":"Rethinking the Role of Prompting Strategies in LLM Test-Time Scaling: A Perspective of Probability Theory","date":"2025-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MraDonkey/rethinking_prompting","path":"model.py","file_url":"https://github.com/MraDonkey/rethinking_prompting/blob/HEAD/model.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8686c0a944293a75","mcp_get_code":{"code_sha256":"8686c0a944293a75"}},{"arxiv_id":"2503.07459","paper":"/paper/medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gersteinlab/medagents-benchmark","path":"baselines/MedPrompt/cot.py","file_url":"https://github.com/gersteinlab/medagents-benchmark/blob/HEAD/baselines/MedPrompt/cot.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"22aed9a263e00947","mcp_get_code":{"code_sha256":"22aed9a263e00947"}},{"arxiv_id":"2503.07459","paper":"/paper/medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gersteinlab/medagents-benchmark","path":"baselines/MedPrompt/cot_sc.py","file_url":"https://github.com/gersteinlab/medagents-benchmark/blob/HEAD/baselines/MedPrompt/cot_sc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cb99011218fa888b","mcp_get_code":{"code_sha256":"cb99011218fa888b"}},{"arxiv_id":"2502.12215","paper":"/paper/revisiting-the-test-time-scaling-of-o1-like","title":"Revisiting the Test-Time Scaling of o1-like Models: Do they Truly Possess Test-Time Scaling Capabilities?","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhiYuanZeng/test-time-scaling-eval","path":"utils.py","file_url":"https://github.com/ZhiYuanZeng/test-time-scaling-eval/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5e704152b8c3311d","mcp_get_code":{"code_sha256":"5e704152b8c3311d"}},{"arxiv_id":"2411.05383","paper":"/paper/towards-low-resource-harmful-meme-detection","title":"Towards Low-Resource Harmful Meme Detection with LMM Agents","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jianzhao-huang/lorehm","path":"utils/utils.py","file_url":"https://github.com/jianzhao-huang/lorehm/blob/HEAD/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c61e5f3d725c152a","mcp_get_code":{"code_sha256":"c61e5f3d725c152a"}},{"arxiv_id":"2410.19546","paper":"/paper/bongard-in-wonderland-visual-puzzles-that","title":"Bongard in Wonderland: Visual Puzzles that Still Make AI Go Mad?","date":"2024-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ml-research/bongard-in-wonderland","path":"experiments/evaluate/bp_llm_judge.py","file_url":"https://github.com/ml-research/bongard-in-wonderland/blob/HEAD/experiments/evaluate/bp_llm_judge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4e1b6d4d2b1f83ce","mcp_get_code":{"code_sha256":"4e1b6d4d2b1f83ce"}},{"arxiv_id":"2410.19546","paper":"/paper/bongard-in-wonderland-visual-puzzles-that","title":"Bongard in Wonderland: Visual Puzzles that Still Make AI Go Mad?","date":"2024-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ml-research/bongard-in-wonderland","path":"experiments/evaluate/bp_llm_judge_human.py","file_url":"https://github.com/ml-research/bongard-in-wonderland/blob/HEAD/experiments/evaluate/bp_llm_judge_human.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e59329369ae20f74","mcp_get_code":{"code_sha256":"e59329369ae20f74"}},{"arxiv_id":"2410.16676","paper":"/paper/improving-causal-reasoning-in-large-language","title":"Improving Causal Reasoning in Large Language Models: A Survey","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chendl02/awesome-llm-causal-reasoning","path":"src/eval_all.py","file_url":"https://github.com/chendl02/awesome-llm-causal-reasoning/blob/HEAD/src/eval_all.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"41f8cc815c276ae9","mcp_get_code":{"code_sha256":"41f8cc815c276ae9"}},{"arxiv_id":"2410.12971","paper":"/paper/self-pluralising-culture-alignment-for-large","title":"Self-Pluralising Culture Alignment for Large Language Models","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shaoyangxu/culturespa","path":"result_analysis_run_3.py","file_url":"https://github.com/shaoyangxu/culturespa/blob/HEAD/result_analysis_run_3.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9411f538a5d9a28b","mcp_get_code":{"code_sha256":"9411f538a5d9a28b"}},{"arxiv_id":"2408.13654","paper":"/paper/symbolic-working-memory-enhances-language","title":"Symbolic Working Memory Enhances Language Models for Complex Rule Application","date":"2024-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"siyuanwangw/ruleapplication","path":"Src/evaluation_utils.py","file_url":"https://github.com/siyuanwangw/ruleapplication/blob/HEAD/Src/evaluation_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1675e6f35d4f8ba4","mcp_get_code":{"code_sha256":"1675e6f35d4f8ba4"}},{"arxiv_id":"2404.10308","paper":"/paper/hierarchical-context-merging-better-long","title":"Hierarchical Context Merging: Better Long Context Understanding for Pre-trained LLMs","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alinlab/HOMER","path":"src/passkey.py","file_url":"https://github.com/alinlab/HOMER/blob/HEAD/src/passkey.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"924e90912400c67d","mcp_get_code":{"code_sha256":"924e90912400c67d"}},{"arxiv_id":"2402.18496","paper":"/paper/language-models-represent-beliefs-of-self-and","title":"Language Models Represent Beliefs of Self and Others","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"walter0807/repbelief","path":"evaluate_conditions.py","file_url":"https://github.com/walter0807/repbelief/blob/HEAD/evaluate_conditions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6b12dca756d315b0","mcp_get_code":{"code_sha256":"6b12dca756d315b0"}},{"arxiv_id":"2402.13093","paper":"/paper/event-level-knowledge-editing","title":"Event-level Knowledge Editing","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-keg/event-level-knowledge-editing","path":"overall_event_level_metric.py","file_url":"https://github.com/thu-keg/event-level-knowledge-editing/blob/HEAD/overall_event_level_metric.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"da8cb85a2c5ca2a6","mcp_get_code":{"code_sha256":"da8cb85a2c5ca2a6"}},{"arxiv_id":"2402.10980","paper":"/paper/chemreasoner-heuristic-search-over-a-large","title":"ChemReasoner: Heuristic Search over a Large Language Model's Knowledge Space using Quantum-Chemical Feedback","date":"2024-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pnnl/chemreasoner","path":"src/search/state/reasoner_state.py","file_url":"https://github.com/pnnl/chemreasoner/blob/HEAD/src/search/state/reasoner_state.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"8ad4e051476a3e90","mcp_get_code":{"code_sha256":"8ad4e051476a3e90"}},{"arxiv_id":"2312.14890","paper":"/paper/nphardeval-dynamic-benchmark-on-reasoning","title":"NPHardEval: Dynamic Benchmark on Reasoning Ability of Large Language Models via Complexity Classes","date":"2023-12-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"casmlab/nphardeval","path":"Close/check/check_hard_GCP.py","file_url":"https://github.com/casmlab/nphardeval/blob/HEAD/Close/check/check_hard_GCP.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"78998436efd67437","mcp_get_code":{"code_sha256":"78998436efd67437"}},{"arxiv_id":"2311.01767","paper":"/paper/pptc-benchmark-evaluating-large-language","title":"PPTC Benchmark: Evaluating Large Language Models for PowerPoint Task Completion","date":"2023-11-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gydpku/pptc","path":"src/content_selection.py","file_url":"https://github.com/gydpku/pptc/blob/HEAD/src/content_selection.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"acccc24d18f3e8af","mcp_get_code":{"code_sha256":"acccc24d18f3e8af"}},{"arxiv_id":"2310.04678","paper":"/paper/doris-mae-scientific-document-retrieval-using","title":"DORIS-MAE: Scientific Document Retrieval using Multi-level Aspect-based Queries","date":"2023-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Real-Doris-Mae/Doris-Mae-Dataset","path":"gpt_annotation/annotation_util.py","file_url":"https://github.com/Real-Doris-Mae/Doris-Mae-Dataset/blob/HEAD/gpt_annotation/annotation_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"cbaf9acd60451d0f","mcp_get_code":{"code_sha256":"cbaf9acd60451d0f"}},{"arxiv_id":"2310.02124","paper":"/paper/exploring-collaboration-mechanisms-for-llm","title":"Exploring Collaboration Mechanisms for LLM Agents: A Social Psychology View","date":"2023-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zjunlp/machinesom","path":"src/evaluate.py","file_url":"https://github.com/zjunlp/machinesom/blob/HEAD/src/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2ee705a8231381c4","mcp_get_code":{"code_sha256":"2ee705a8231381c4"}},{"arxiv_id":"2306.02546","paper":"/paper/lmpa-improving-decompilation-by-synergy-of","title":"Symbol Preference Aware Generative Models for Recovering Variable Names from Stripped Binary","date":null,"month_inferred_from_arxiv_id":"2023-06","title_source":"archive","repo":"xz-x/gennm-ndss-ae","path":"inference/infer_vllm.py","file_url":"https://github.com/xz-x/gennm-ndss-ae/blob/HEAD/inference/infer_vllm.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1f5426ac10ec0aad","mcp_get_code":{"code_sha256":"1f5426ac10ec0aad"}},{"arxiv_id":"2024.findings-acl.514","paper":null,"title":"arXiv:2024.findings-acl.514","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"gydpku/PPTC","path":"src/content_selection.py","file_url":"https://github.com/gydpku/PPTC/blob/HEAD/src/content_selection.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"acccc24d18f3e8af","mcp_get_code":{"code_sha256":"acccc24d18f3e8af"}},{"arxiv_id":"2024.acl-long.225","paper":null,"title":"arXiv:2024.acl-long.225","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"casmlab/NPHardEval","path":"Close/check/check_hard_GCP.py","file_url":"https://github.com/casmlab/NPHardEval/blob/HEAD/Close/check/check_hard_GCP.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"78998436efd67437","mcp_get_code":{"code_sha256":"78998436efd67437"}}]}