{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/call-gpt","entry":"call_gpt","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":17,"n_papers_ran":6,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":17,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":19,"n_places_pointer_only":11,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":3,"unverified":13},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.01204","paper":"/paper/arxiv-2606-01204","title":"Implicit Geographic Inference in LLM Medical Triage: Language-Driven Disparities in Emergency Recommendations","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"wongqihan/ai-behavioral-experiments","path":"gender-age-triage/run_multimodel.py","file_url":"https://github.com/wongqihan/ai-behavioral-experiments/blob/HEAD/gender-age-triage/run_multimodel.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e3d170768035e5fe","mcp_get_code":{"code_sha256":"e3d170768035e5fe"}},{"arxiv_id":"2601.04577","paper":"/paper/arxiv-2601-04577","title":"Sci-Reasoning: A Dataset Decoding AI Innovation Patterns","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"AmberLJC/Sci-Reasoning","path":"thinking_patterns_llm_analysis/code/pattern_analyzer.py","file_url":"https://github.com/AmberLJC/Sci-Reasoning/blob/HEAD/thinking_patterns_llm_analysis/code/pattern_analyzer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5206dc3f5d475fe4","mcp_get_code":{"code_sha256":"5206dc3f5d475fe4"}},{"arxiv_id":"2507.06167","paper":"/paper/skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"0a19144295a700b0","mcp_get_code":{"code_sha256":"0a19144295a700b0"}},{"arxiv_id":"2506.14205","paper":"/paper/agentsynth-scalable-task-generation-for","title":"AgentSynth: Scalable Task Generation for Generalist Computer-Use Agents","date":"2025-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunblaze-ucb/agentsynth","path":"agentsynth/utils.py","file_url":"https://github.com/sunblaze-ucb/agentsynth/blob/HEAD/agentsynth/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f8c9174688291d66","mcp_get_code":{"code_sha256":"f8c9174688291d66"}},{"arxiv_id":"2505.20292","paper":"/paper/opens2v-nexus-a-detailed-benchmark-and","title":"OpenS2V-Nexus: A Detailed Benchmark and Million-Scale Dataset for Subject-to-Video Generation","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PKU-YuanGroup/OpenS2V-Nexus","path":"data_process/step4-1_get_tag_api.py","file_url":"https://github.com/PKU-YuanGroup/OpenS2V-Nexus/blob/HEAD/data_process/step4-1_get_tag_api.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b9566bb371805bec","mcp_get_code":{"code_sha256":"b9566bb371805bec"}},{"arxiv_id":"2504.16656","paper":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement","title":"Skywork R1V2: Multimodal Hybrid Reinforcement Learning for Reasoning","date":"2025-04-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"0a19144295a700b0","mcp_get_code":{"code_sha256":"0a19144295a700b0"}},{"arxiv_id":"2504.05599","paper":"/paper/skywork-r1v-pioneering-multimodal-reasoning","title":"Skywork R1V: Pioneering Multimodal Reasoning with Chain-of-Thought","date":"2025-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SkyworkAI/Skywork-R1V","path":"eval/EMMA/evaluation/evaluate.py","file_url":"https://github.com/SkyworkAI/Skywork-R1V/blob/HEAD/eval/EMMA/evaluation/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0a19144295a700b0","mcp_get_code":{"code_sha256":"0a19144295a700b0"}},{"arxiv_id":"2412.13178","paper":"/paper/safeagentbench-a-benchmark-for-safe-task","title":"SafeAgentBench: A Benchmark for Safe Task Planning of Embodied LLM Agents","date":"2024-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shengyin1224/safeagentbench","path":"evaluator/abstract_evaluate.py","file_url":"https://github.com/shengyin1224/safeagentbench/blob/HEAD/evaluator/abstract_evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"004ef77a803ae087","mcp_get_code":{"code_sha256":"004ef77a803ae087"}},{"arxiv_id":"2412.13178","paper":"/paper/safeagentbench-a-benchmark-for-safe-task","title":"SafeAgentBench: A Benchmark for Safe Task Planning of Embodied LLM Agents","date":"2024-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shengyin1224/safeagentbench","path":"evaluator/detail_evaluate.py","file_url":"https://github.com/shengyin1224/safeagentbench/blob/HEAD/evaluator/detail_evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2f61c26743e5287d","mcp_get_code":{"code_sha256":"2f61c26743e5287d"}},{"arxiv_id":"2412.13178","paper":"/paper/safeagentbench-a-benchmark-for-safe-task","title":"SafeAgentBench: A Benchmark for Safe Task Planning of Embodied LLM Agents","date":"2024-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shengyin1224/safeagentbench","path":"evaluator/long_horizon_evaluate.py","file_url":"https://github.com/shengyin1224/safeagentbench/blob/HEAD/evaluator/long_horizon_evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cf64676d260a87c6","mcp_get_code":{"code_sha256":"cf64676d260a87c6"}},{"arxiv_id":"2411.02937","paper":"/paper/benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba-nlp/omnisearch","path":"src/Omnisearch_gpt/llm_config.py","file_url":"https://github.com/alibaba-nlp/omnisearch/blob/HEAD/src/Omnisearch_gpt/llm_config.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a8a9cae0a25d828f","mcp_get_code":{"code_sha256":"a8a9cae0a25d828f"}},{"arxiv_id":"2410.09584","paper":"/paper/toward-general-instruction-following","title":"Toward General Instruction-Following Alignment for Retrieval-Augmented Generation","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongguanting/FollowRAG","path":"FollowRAG/utils/call_llm.py","file_url":"https://github.com/dongguanting/FollowRAG/blob/HEAD/FollowRAG/utils/call_llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"30de2ad5d6678c9d","mcp_get_code":{"code_sha256":"30de2ad5d6678c9d"}},{"arxiv_id":"2410.02507","paper":"/paper/can-large-language-models-grasp-legal","title":"Can Large Language Models Grasp Legal Theories? Enhance Legal Reasoning with Insights from Multi-Agent Collaboration","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuanwk99/malr","path":"MALR/utils/model.py","file_url":"https://github.com/yuanwk99/malr/blob/HEAD/MALR/utils/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5f4ee4fcaeb3d795","mcp_get_code":{"code_sha256":"5f4ee4fcaeb3d795"}},{"arxiv_id":"2409.12953","paper":"/paper/journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"journeybench/journeybench","path":"automatic-qa-generator/chat/call_gpt.py","file_url":"https://github.com/journeybench/journeybench/blob/HEAD/automatic-qa-generator/chat/call_gpt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"b989da394071b3df","mcp_get_code":{"code_sha256":"b989da394071b3df"}},{"arxiv_id":"2407.07087","paper":"/paper/copybench-measuring-literal-and-non-literal","title":"CopyBench: Measuring Literal and Non-Literal Reproduction of Copyright-Protected Text in Language Model Generation","date":"2024-07-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chentong0/copy-bench","path":"src/utils/openai_lm.py","file_url":"https://github.com/chentong0/copy-bench/blob/HEAD/src/utils/openai_lm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"87aa8dba7017b976","mcp_get_code":{"code_sha256":"87aa8dba7017b976"}},{"arxiv_id":"2310.00305","paper":"/paper/towards-llm-based-fact-verification-on-news","title":"Towards LLM-based Fact Verification on News Claims with a Hierarchical Step-by-Step Prompting Method","date":"2023-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jadecurl/hiss","path":"HiSS.py","file_url":"https://github.com/jadecurl/hiss/blob/HEAD/HiSS.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2a1c9ffdc3710e45","mcp_get_code":{"code_sha256":"2a1c9ffdc3710e45"}},{"arxiv_id":"2309.06275","paper":"/paper/2309-06275","title":"Re-Reading Improves Reasoning in Large Language Models","date":"2023-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tebmer/rereading-llm-reasoning","path":"pal/core/backend.py","file_url":"https://github.com/tebmer/rereading-llm-reasoning/blob/HEAD/pal/core/backend.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d2417be2799a98fe","mcp_get_code":{"code_sha256":"d2417be2799a98fe"}},{"arxiv_id":"2304.13007","paper":"/paper/answering-questions-by-meta-reasoning-over","title":"Answering Questions by Meta-Reasoning over Multiple Chains of Thought","date":"2023-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"oriyor/reasoning-on-cots","path":"src/opeanai/utils.py","file_url":"https://github.com/oriyor/reasoning-on-cots/blob/HEAD/src/opeanai/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"63cfc8eb0c95d0dc","mcp_get_code":{"code_sha256":"63cfc8eb0c95d0dc"}},{"arxiv_id":"2024.acl-long.438","paper":null,"title":"arXiv:2024.acl-long.438","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Hengrui-Gu/PokeMQA","path":"PokeMQA-turbo_n_edited.py","file_url":"https://github.com/Hengrui-Gu/PokeMQA/blob/HEAD/PokeMQA-turbo_n_edited.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e02b14c5a6ec4bc5","mcp_get_code":{"code_sha256":"e02b14c5a6ec4bc5"}}]}