{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-response","entry":"get_response","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":48,"n_papers_ran":15,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":60,"n_samples_ran":17,"n_samples_fingerprinted":1,"n_places":60,"n_places_pointer_only":28,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":14,"unverified":43},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.12655","paper":"/paper/arxiv-2609-12655","title":"LifeMem: Enabling Lifelong Experience Reuse for LLM Agents","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"BITHLP/LifeMem","path":"api-bank/utils.py","file_url":"https://github.com/BITHLP/LifeMem/blob/HEAD/api-bank/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"831b9407e8cf0755","mcp_get_code":{"code_sha256":"831b9407e8cf0755"}},{"arxiv_id":"2604.25847","paper":"/paper/arxiv-2604-25847","title":"From Soliloquy to Agora: Memory-Enhanced LLM Agents with Decentralized Debate for Optimization Modeling","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"CHIANGEL/Agora-Opt","path":"code/Agora-Opt/src/debate_memory/llm.py","file_url":"https://github.com/CHIANGEL/Agora-Opt/blob/HEAD/code/Agora-Opt/src/debate_memory/llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bc49cf903440df48","mcp_get_code":{"code_sha256":"bc49cf903440df48"}},{"arxiv_id":"2604.13592","paper":"/paper/arxiv-2604-13592","title":"Foresight Optimization for Strategic Reasoning in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"wangjs9/ForesightOptim","path":"competitive_taboo/dialogue_refinement.py","file_url":"https://github.com/wangjs9/ForesightOptim/blob/HEAD/competitive_taboo/dialogue_refinement.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a28f0703801c60e0","mcp_get_code":{"code_sha256":"a28f0703801c60e0"}},{"arxiv_id":"2510.05115","paper":"/paper/arxiv-2510-05115","title":"SAC-Opt: Semantic Anchors for Iterative Correction in Optimization Modeling","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"Forrest-Stone/SAC-Opt","path":"constraint.py","file_url":"https://github.com/Forrest-Stone/SAC-Opt/blob/HEAD/constraint.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2b61f84f66061098","mcp_get_code":{"code_sha256":"2b61f84f66061098"}},{"arxiv_id":"2508.06492","paper":"/paper/arxiv-2508-06492","title":"Effective Training Data Synthesis for Improving MLLM Chart Understanding","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"yuweiyang-anu/ECD","path":"data_generation_pipeline/chart_image_filtering.py","file_url":"https://github.com/yuweiyang-anu/ECD/blob/HEAD/data_generation_pipeline/chart_image_filtering.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cc556a6a1407e0b2","mcp_get_code":{"code_sha256":"cc556a6a1407e0b2"}},{"arxiv_id":"2508.06492","paper":"/paper/arxiv-2508-06492","title":"Effective Training Data Synthesis for Improving MLLM Chart Understanding","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"yuweiyang-anu/ECD","path":"data_generation_pipeline/descriptive_qa_generation.py","file_url":"https://github.com/yuweiyang-anu/ECD/blob/HEAD/data_generation_pipeline/descriptive_qa_generation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c2283b6d56917f7b","mcp_get_code":{"code_sha256":"c2283b6d56917f7b"}},{"arxiv_id":"2505.19037","paper":"/paper/speech-ifeval-evaluating-instruction","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","date":"2025-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kehanlu/speech-ifeval","path":"examples/eval_desta2.py","file_url":"https://github.com/kehanlu/speech-ifeval/blob/HEAD/examples/eval_desta2.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4cad2548f8d71d0b","mcp_get_code":{"code_sha256":"4cad2548f8d71d0b"}},{"arxiv_id":"2505.05505","paper":"/paper/apply-hierarchical-chain-of-generation-to","title":"Apply Hierarchical-Chain-of-Generation to Complex Attributes Text-to-3D Generation","date":"2025-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Wakals/GASCOL","path":"threestudio/gpt/PE.py","file_url":"https://github.com/Wakals/GASCOL/blob/HEAD/threestudio/gpt/PE.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4ade5610ab000535","mcp_get_code":{"code_sha256":"4ade5610ab000535"}},{"arxiv_id":"2503.15234","paper":"/paper/coe-chain-of-explanation-via-automatic-visual","title":"CoE: Chain-of-Explanation via Automatic Visual Concept Circuit Description and Polysemanticity Quantification","date":"2025-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YuWLong666/CoE","path":"closeai.py","file_url":"https://github.com/YuWLong666/CoE/blob/HEAD/closeai.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"43fb79dfb73bea95","mcp_get_code":{"code_sha256":"43fb79dfb73bea95"}},{"arxiv_id":"2502.16690","paper":"/paper/from-text-to-space-mapping-abstract-spatial","title":"From Text to Space: Mapping Abstract Spatial Models in LLMs during a Grid-World Navigation Task","date":"2025-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mneuronico/griw-world-spatial-orientation-task","path":"experiments/fns.py","file_url":"https://github.com/mneuronico/griw-world-spatial-orientation-task/blob/HEAD/experiments/fns.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1380efbe5fa29e71","mcp_get_code":{"code_sha256":"1380efbe5fa29e71"}},{"arxiv_id":"2502.00657","paper":"/paper/llm-safety-alignment-is-divergence-estimation","title":"LLM Safety Alignment is Divergence Estimation in Disguise","date":"2025-02-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rhaldarpurdue/kldo","path":"dataset_generation/compare.py","file_url":"https://github.com/rhaldarpurdue/kldo/blob/HEAD/dataset_generation/compare.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"962f72a6b771258f","mcp_get_code":{"code_sha256":"962f72a6b771258f"}},{"arxiv_id":"2501.00830","paper":"/paper/llm-al-bridging-large-language-models-and","title":"LLM+AL: Bridging Large Language Models and Action Languages for Complex Reasoning about Actions","date":"2025-01-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"azreasoners/llm-al","path":"utils.py","file_url":"https://github.com/azreasoners/llm-al/blob/HEAD/utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"535397c1d5ef3d19","mcp_get_code":{"code_sha256":"535397c1d5ef3d19"}},{"arxiv_id":"2412.06606","paper":"/paper/vulnerability-of-text-matching-in-ml-ai","title":"Vulnerability of Text-Matching in ML/AI Conference Reviewer Assignments to Collusions","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"passionfruit03/reviewer_assignments_vulnerability","path":"attack/gpt_helpers.py","file_url":"https://github.com/passionfruit03/reviewer_assignments_vulnerability/blob/HEAD/attack/gpt_helpers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"f0f623d87b6930c8","mcp_get_code":{"code_sha256":"f0f623d87b6930c8"}},{"arxiv_id":"2411.13504","paper":"/paper/disentangling-memory-and-reasoning-ability-in","title":"Disentangling Memory and Reasoning Ability in Large Language Models","date":"2024-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mingyuj666/disentangling-memory-and-reasoning","path":"load_data/data_agent.py","file_url":"https://github.com/mingyuj666/disentangling-memory-and-reasoning/blob/HEAD/load_data/data_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3842459b68f0f8a0","mcp_get_code":{"code_sha256":"3842459b68f0f8a0"}},{"arxiv_id":"2411.13504","paper":"/paper/disentangling-memory-and-reasoning-ability-in","title":"Disentangling Memory and Reasoning Ability in Large Language Models","date":"2024-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mingyuj666/disentangling-memory-and-reasoning","path":"load_data/data_agent_modify_version.py","file_url":"https://github.com/mingyuj666/disentangling-memory-and-reasoning/blob/HEAD/load_data/data_agent_modify_version.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5187be23c430fe41","mcp_get_code":{"code_sha256":"5187be23c430fe41"}},{"arxiv_id":"2411.13504","paper":"/paper/disentangling-memory-and-reasoning-ability-in","title":"Disentangling Memory and Reasoning Ability in Large Language Models","date":"2024-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mingyuj666/disentangling-memory-and-reasoning","path":"load_data/data_agent_order_modify_version.py","file_url":"https://github.com/mingyuj666/disentangling-memory-and-reasoning/blob/HEAD/load_data/data_agent_order_modify_version.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6571c6d3a61d5838","mcp_get_code":{"code_sha256":"6571c6d3a61d5838"}},{"arxiv_id":"2410.22587","paper":"/paper/toxicity-of-the-commons-curating-open-source","title":"Toxicity of the Commons: Curating Open-Source Pre-Training Data","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Pleias/toxic-commons","path":"src/2.1_create_annotations.py","file_url":"https://github.com/Pleias/toxic-commons/blob/HEAD/src/2.1_create_annotations.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"68a1776789f90465","mcp_get_code":{"code_sha256":"68a1776789f90465"}},{"arxiv_id":"2410.16251","paper":"/paper/can-knowledge-editing-really-correct","title":"Can Knowledge Editing Really Correct Hallucinations?","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-editing/HalluEditBench","path":"code/eval_hallu.py","file_url":"https://github.com/llm-editing/HalluEditBench/blob/HEAD/code/eval_hallu.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e82d0dc93553559b","mcp_get_code":{"code_sha256":"e82d0dc93553559b"}},{"arxiv_id":"2410.16251","paper":"/paper/can-knowledge-editing-really-correct","title":"Can Knowledge Editing Really Correct Hallucinations?","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-editing/HalluEditBench","path":"code/hallucination_editor.py","file_url":"https://github.com/llm-editing/HalluEditBench/blob/HEAD/code/hallucination_editor.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c345aa339c992639","mcp_get_code":{"code_sha256":"c345aa339c992639"}},{"arxiv_id":"2410.16251","paper":"/paper/can-knowledge-editing-really-correct","title":"Can Knowledge Editing Really Correct Hallucinations?","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-editing/HalluEditBench","path":"code/util.py","file_url":"https://github.com/llm-editing/HalluEditBench/blob/HEAD/code/util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"07845ab4336c94fa","mcp_get_code":{"code_sha256":"07845ab4336c94fa"}},{"arxiv_id":"2410.10799","paper":"/paper/towards-foundation-models-for-3d-vision-how","title":"Towards Foundation Models for 3D Vision: How Close Are We?","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-vl/uniqa-3d","path":"LLM_evaluations/clevr_vqa/generate_gpt_response.py","file_url":"https://github.com/princeton-vl/uniqa-3d/blob/HEAD/LLM_evaluations/clevr_vqa/generate_gpt_response.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"f609b882d3318e8a","mcp_get_code":{"code_sha256":"f609b882d3318e8a"}},{"arxiv_id":"2410.10799","paper":"/paper/towards-foundation-models-for-3d-vision-how","title":"Towards Foundation Models for 3D Vision: How Close Are We?","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-vl/uniqa-3d","path":"LLM_evaluations/relative_camera_pose/generate_gpt4v_response.py","file_url":"https://github.com/princeton-vl/uniqa-3d/blob/HEAD/LLM_evaluations/relative_camera_pose/generate_gpt4v_response.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"864f8f7c8a83f0b3","mcp_get_code":{"code_sha256":"864f8f7c8a83f0b3"}},{"arxiv_id":"2410.10799","paper":"/paper/towards-foundation-models-for-3d-vision-how","title":"Towards Foundation Models for 3D Vision: How Close Are We?","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-vl/uniqa-3d","path":"LLM_evaluations/relative_depth/generate_gpt4v_response.py","file_url":"https://github.com/princeton-vl/uniqa-3d/blob/HEAD/LLM_evaluations/relative_depth/generate_gpt4v_response.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"42bf1823205a9ad2","mcp_get_code":{"code_sha256":"42bf1823205a9ad2"}},{"arxiv_id":"2410.10626","paper":"/paper/efficiently-democratizing-medical-llms-for-50","title":"Efficiently Democratizing Medical LLMs for 50 Languages via a Mixture of Language Family Experts","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/apollomoe","path":"src/eval/eval_gemma.py","file_url":"https://github.com/freedomintelligence/apollomoe/blob/HEAD/src/eval/eval_gemma.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"56889f7eefb6cd73","mcp_get_code":{"code_sha256":"56889f7eefb6cd73"}},{"arxiv_id":"2410.08197","paper":"/paper/from-exploration-to-mastery-enabling-llms-to","title":"From Exploration to Mastery: Enabling LLMs to Master Tools via Self-Driven Interactions","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"quchangle1/DRAFT","path":"Inference_DFSDT.py","file_url":"https://github.com/quchangle1/DRAFT/blob/HEAD/Inference_DFSDT.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e247a2919937c8b6","mcp_get_code":{"code_sha256":"e247a2919937c8b6"}},{"arxiv_id":"2409.15477","paper":"/paper/mediconfusion-can-you-trust-your-ai","title":"MediConfusion: Can you trust your AI radiologist? Probing the reliability of multimodal medical foundation models","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AIF4S/MediConfusion","path":"Models/gpt.py","file_url":"https://github.com/AIF4S/MediConfusion/blob/HEAD/Models/gpt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"40fb3b39651a7e1e","mcp_get_code":{"code_sha256":"40fb3b39651a7e1e"}},{"arxiv_id":"2409.15154","paper":"/paper/rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qing-yuan233/RMCBench","path":"script/evaluate.py","file_url":"https://github.com/qing-yuan233/RMCBench/blob/HEAD/script/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"bdd86ff0c494eaec","mcp_get_code":{"code_sha256":"bdd86ff0c494eaec"}},{"arxiv_id":"2409.15154","paper":"/paper/rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qing-yuan233/RMCBench","path":"script/run_gpt_llm.py","file_url":"https://github.com/qing-yuan233/RMCBench/blob/HEAD/script/run_gpt_llm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"162109372f2d4acb","mcp_get_code":{"code_sha256":"162109372f2d4acb"}},{"arxiv_id":"2409.15154","paper":"/paper/rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qing-yuan233/RMCBench","path":"script/run_open_llm.py","file_url":"https://github.com/qing-yuan233/RMCBench/blob/HEAD/script/run_open_llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a04cf0f788099abd","mcp_get_code":{"code_sha256":"a04cf0f788099abd"}},{"arxiv_id":"2407.20224","paper":"/paper/can-editing-llms-inject-harm","title":"Can Editing LLMs Inject Harm?","date":"2024-07-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-editing/editing-attack","path":"code/editor_new_eval.py","file_url":"https://github.com/llm-editing/editing-attack/blob/HEAD/code/editor_new_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"446fc0f34e412c18","mcp_get_code":{"code_sha256":"446fc0f34e412c18"}},{"arxiv_id":"2407.16637","paper":"/paper/course-correction-safety-alignment-using","title":"Course-Correction: Safety Alignment Using Synthetic Preferences","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pillowsofwind/course-correction","path":"eval/eval_data.py","file_url":"https://github.com/pillowsofwind/course-correction/blob/HEAD/eval/eval_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b7dd78298a1e253","mcp_get_code":{"code_sha256":"0b7dd78298a1e253"}},{"arxiv_id":"2406.11678","paper":"/paper/tourrank-utilizing-large-language-models-for","title":"TourRank: Utilizing Large Language Models for Documents Ranking with a Tournament-Inspired Strategy","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chenyiqun/TourRank","path":"TourRank_multiprocessing.py","file_url":"https://github.com/chenyiqun/TourRank/blob/HEAD/TourRank_multiprocessing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"74aeaeb3787e659c","mcp_get_code":{"code_sha256":"74aeaeb3787e659c"}},{"arxiv_id":"2405.19119","paper":"/paper/can-graph-learning-improve-task-planning","title":"Can Graph Learning Improve Planning in LLM-based Agents?","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wxxshirley/gnn4taskplan","path":"trainfree/direct.py","file_url":"https://github.com/wxxshirley/gnn4taskplan/blob/HEAD/trainfree/direct.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fad6e5fad73b7b31","mcp_get_code":{"code_sha256":"fad6e5fad73b7b31"}},{"arxiv_id":"2405.19119","paper":"/paper/can-graph-learning-improve-task-planning","title":"Can Graph Learning Improve Planning in LLM-based Agents?","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wxxshirley/gnn4taskplan","path":"trainfree/direct_diffprompt.py","file_url":"https://github.com/wxxshirley/gnn4taskplan/blob/HEAD/trainfree/direct_diffprompt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d85fa758ae62b493","mcp_get_code":{"code_sha256":"d85fa758ae62b493"}},{"arxiv_id":"2405.14702","paper":"/paper/g3-an-effective-and-adaptive-framework-for","title":"G3: An Effective and Adaptive Framework for Worldwide Geolocalization Using Large Multi-Modality Models","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"applied-machine-learning-lab/g3","path":"llm_predict.py","file_url":"https://github.com/applied-machine-learning-lab/g3/blob/HEAD/llm_predict.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"14ab877eaf533dfb","mcp_get_code":{"code_sha256":"14ab877eaf533dfb"}},{"arxiv_id":"2405.14702","paper":"/paper/g3-an-effective-and-adaptive-framework-for","title":"G3: An Effective and Adaptive Framework for Worldwide Geolocalization Using Large Multi-Modality Models","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"applied-machine-learning-lab/g3","path":"llm_predict_hf.py","file_url":"https://github.com/applied-machine-learning-lab/g3/blob/HEAD/llm_predict_hf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a32ec97a24a3f7e6","mcp_get_code":{"code_sha256":"a32ec97a24a3f7e6"}},{"arxiv_id":"2405.14365","paper":"/paper/jiuzhang3-0-efficiently-improving","title":"JiuZhang3.0: Efficiently Improving Mathematical Reasoning by Training Small Data Synthesis Models","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/jiuzhang3.0","path":"eval/math_eval_openai.py","file_url":"https://github.com/rucaibox/jiuzhang3.0/blob/HEAD/eval/math_eval_openai.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e6e097bbc1e5c6d4","mcp_get_code":{"code_sha256":"e6e097bbc1e5c6d4"}},{"arxiv_id":"2404.01197","paper":"/paper/getting-it-right-improving-spatial","title":"Getting it Right: Improving Spatial Consistency in Text-to-Image Models","date":"2024-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SPRIGHT-T2I/SPRIGHT","path":"eval/gpt4/eval_with_gpt4.py","file_url":"https://github.com/SPRIGHT-T2I/SPRIGHT/blob/HEAD/eval/gpt4/eval_with_gpt4.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b153819851508e8f","mcp_get_code":{"code_sha256":"b153819851508e8f"}},{"arxiv_id":"2403.10667","paper":"/paper/towards-unified-multi-modal-personalization","title":"Towards Unified Multi-Modal Personalization: Large Vision-Language Models for Generative Recommendation and Beyond","date":"2024-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weitianxin/UniMP","path":"UniMP/pipeline/eval/benchmark_otter.py","file_url":"https://github.com/weitianxin/UniMP/blob/HEAD/UniMP/pipeline/eval/benchmark_otter.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8b8566f333cb4974","mcp_get_code":{"code_sha256":"8b8566f333cb4974"}},{"arxiv_id":"2402.11821","paper":"/paper/microstructures-and-accuracy-of-graph-recall","title":"Microstructures and Accuracy of Graph Recall by Large Language Models","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"abel0828/llm-graph-recall","path":"network_recall.py","file_url":"https://github.com/abel0828/llm-graph-recall/blob/HEAD/network_recall.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"190871a24a3d1816","mcp_get_code":{"code_sha256":"190871a24a3d1816"}},{"arxiv_id":"2402.11622","paper":"/paper/logical-closed-loop-uncovering-object","title":"Logical Closed Loop: Uncovering Object Hallucinations in Large Vision-Language Models","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hyperwjf/logiccheckgpt","path":"logiccheckgpt/check_llava.py","file_url":"https://github.com/hyperwjf/logiccheckgpt/blob/HEAD/logiccheckgpt/check_llava.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c94c0823ad207c14","mcp_get_code":{"code_sha256":"c94c0823ad207c14"}},{"arxiv_id":"2402.11349","paper":"/paper/tasks-that-language-models-don-t-learn","title":"Language Models Don't Learn the Physical Manifestation of Language","date":"2024-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"brucewlee/h-test","path":"utils.py","file_url":"https://github.com/brucewlee/h-test/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0e73b3bcb07d374d","mcp_get_code":{"code_sha256":"0e73b3bcb07d374d"}},{"arxiv_id":"2402.03244","paper":"/paper/skill-set-optimization-reinforcing-language","title":"Skill Set Optimization: Reinforcing Language Model Behavior via Transferable Skills","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/sso","path":"sso/llm/gpt.py","file_url":"https://github.com/allenai/sso/blob/HEAD/sso/llm/gpt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3ccfa89ad4d945bd","mcp_get_code":{"code_sha256":"3ccfa89ad4d945bd"}},{"arxiv_id":"2312.11444","paper":"/paper/an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"neulab/gemini-benchmark","path":"benchmarking/Code/run_code.py","file_url":"https://github.com/neulab/gemini-benchmark/blob/HEAD/benchmarking/Code/run_code.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"78ca0bb4c28987ab","mcp_get_code":{"code_sha256":"78ca0bb4c28987ab"}},{"arxiv_id":"2312.06867","paper":"/paper/get-an-a-in-math-progressive-rectification","title":"Get an A in Math: Progressive Rectification Prompting","date":"2023-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wzy6642/PRP","path":"code/utils.py","file_url":"https://github.com/wzy6642/PRP/blob/HEAD/code/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"af57429892b8a7dd","mcp_get_code":{"code_sha256":"af57429892b8a7dd"}},{"arxiv_id":"2311.09774","paper":"/paper/huatuogpt-ii-one-stage-training-for-medical","title":"HuatuoGPT-II, One-stage Training for Medical Adaption of LLMs","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/huatuogpt-ii","path":"evaluation/eval_qa.py","file_url":"https://github.com/freedomintelligence/huatuogpt-ii/blob/HEAD/evaluation/eval_qa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0cc6223401defc70","mcp_get_code":{"code_sha256":"0cc6223401defc70"}},{"arxiv_id":"2310.17342","paper":"/paper/act-sql-in-context-learning-for-text-to-sql","title":"ACT-SQL: In-Context Learning for Text-to-SQL with Automatically-Generated Chain-of-Thought","date":"2023-10-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"x-lance/text2sql-gpt","path":"util/gpt.py","file_url":"https://github.com/x-lance/text2sql-gpt/blob/HEAD/util/gpt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"83512c75d8835ed6","mcp_get_code":{"code_sha256":"83512c75d8835ed6"}},{"arxiv_id":"2310.09550","paper":"/paper/can-large-language-model-comprehend-ancient","title":"Can Large Language Model Comprehend Ancient Chinese? A Preliminary Test on ACLUE","date":"2023-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"isen-zhang/aclue","path":"src/chatgpt.py","file_url":"https://github.com/isen-zhang/aclue/blob/HEAD/src/chatgpt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0df614fff2114f23","mcp_get_code":{"code_sha256":"0df614fff2114f23"}},{"arxiv_id":"2310.09550","paper":"/paper/can-large-language-model-comprehend-ancient","title":"Can Large Language Model Comprehend Ancient Chinese? A Preliminary Test on ACLUE","date":"2023-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"isen-zhang/aclue","path":"src/gpt4.py","file_url":"https://github.com/isen-zhang/aclue/blob/HEAD/src/gpt4.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d0da2ec45f606dab","mcp_get_code":{"code_sha256":"d0da2ec45f606dab"}},{"arxiv_id":"2310.05492","paper":"/paper/how-abilities-in-large-language-models-are","title":"How Abilities in Large Language Models are Affected by Supervised Fine-tuning Data Composition","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangrongsheng/caregpt","path":"ChatGPT/7_webui.py","file_url":"https://github.com/wangrongsheng/caregpt/blob/HEAD/ChatGPT/7_webui.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"ee526ac42e46dde5","mcp_get_code":{"code_sha256":"ee526ac42e46dde5"}},{"arxiv_id":"2308.08973","paper":"/paper/beam-retrieval-general-end-to-end-retrieval","title":"End-to-End Beam Retrieval for Multi-Hop Question Answering","date":"2023-08-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"canghongjian/beam_retriever","path":"llm_exp_long.py","file_url":"https://github.com/canghongjian/beam_retriever/blob/HEAD/llm_exp_long.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0a8f1af973c82ca0","mcp_get_code":{"code_sha256":"0a8f1af973c82ca0"}},{"arxiv_id":"2308.03279","paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"universal-ner/universal-ner","path":"src/utils.py","file_url":"https://github.com/universal-ner/universal-ner/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fbd915d7d3d14137","mcp_get_code":{"code_sha256":"fbd915d7d3d14137"}},{"arxiv_id":"2307.07696","paper":"/paper/coupling-large-language-models-with-logic","title":"Coupling Large Language Models with Logic Programming for Robust and General Reasoning from Text","date":"2023-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"azreasoners/llm-asp","path":"bAbI/babi_utils.py","file_url":"https://github.com/azreasoners/llm-asp/blob/HEAD/bAbI/babi_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"087cdfe304604e4b","mcp_get_code":{"code_sha256":"087cdfe304604e4b"}},{"arxiv_id":"2210.11689","paper":"/paper/sling-sino-linguistic-evaluation-of-large","title":"SLING: Sino Linguistic Evaluation of Large Language Models","date":"2022-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yixiao-Song/SLING_Data_Code","path":"SLING_Code/gpt3_sling.py","file_url":"https://github.com/Yixiao-Song/SLING_Data_Code/blob/HEAD/SLING_Code/gpt3_sling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2eff6ef3426dba90","mcp_get_code":{"code_sha256":"2eff6ef3426dba90"}},{"arxiv_id":"2106.07597","paper":"/paper/mlperf-tiny-benchmark","title":"MLPerf Tiny Benchmark","date":"2021-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mlcommons/tiny","path":"benchmark/runner/stream_wav_uart.py","file_url":"https://github.com/mlcommons/tiny/blob/HEAD/benchmark/runner/stream_wav_uart.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ffecdb9ac0a5c52f","mcp_get_code":{"code_sha256":"ffecdb9ac0a5c52f"}},{"arxiv_id":"1904.02357","paper":"/paper/plan-write-and-revise-an-interactive-system","title":"Plan, Write, and Revise: an Interactive System for Open-Domain Story Generation","date":"2019-04-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seraphinatarrant/plan-write-revise","path":"server/web_server.py","file_url":"https://github.com/seraphinatarrant/plan-write-revise/blob/HEAD/server/web_server.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6ad0d0e2f5fcc46","mcp_get_code":{"code_sha256":"f6ad0d0e2f5fcc46"}},{"arxiv_id":"openreview_9aU4vrHPKD","paper":null,"title":"arXiv:openreview_9aU4vrHPKD","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Octobrist/CoPE","path":"src/models_fschat.py","file_url":"https://github.com/Octobrist/CoPE/blob/HEAD/src/models_fschat.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"45b4efeb901105d6","mcp_get_code":{"code_sha256":"45b4efeb901105d6"}},{"arxiv_id":"ijcai2024_0687","paper":null,"title":"arXiv:ijcai2024_0687","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"zjunlp/FactCHD","path":"data_generate/openai_service.py","file_url":"https://github.com/zjunlp/FactCHD/blob/HEAD/data_generate/openai_service.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4fdbaf2046abb43d","mcp_get_code":{"code_sha256":"4fdbaf2046abb43d"}},{"arxiv_id":"2025.acl-long.806","paper":null,"title":"arXiv:2025.acl-long.806","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"lzw108/RAEmoLLM","path":"indexconstruct/postprocess_label.py","file_url":"https://github.com/lzw108/RAEmoLLM/blob/HEAD/indexconstruct/postprocess_label.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"09198e1b26cec243","mcp_get_code":{"code_sha256":"09198e1b26cec243"}},{"arxiv_id":"2024.findings-acl.485","paper":null,"title":"arXiv:2024.findings-acl.485","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"CLINEEK/ELAGENT","path":"src/api/concurrent_chat.py","file_url":"https://github.com/CLINEEK/ELAGENT/blob/HEAD/src/api/concurrent_chat.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0a6fc032171a0204","mcp_get_code":{"code_sha256":"0a6fc032171a0204"}}]}