{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-prompt","entry":"load_prompt","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":25,"n_papers_ran":12,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":24,"n_samples_ran":12,"n_samples_fingerprinted":0,"n_places":25,"n_places_pointer_only":17,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":5,"ran_fixture":0,"ran":7,"unverified":12},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.14796","paper":"/paper/arxiv-2609-14796","title":"AI Persuasion as a Threat to Human Control","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"AlignmentResearch/puc","path":"prompts/loader.py","file_url":"https://github.com/AlignmentResearch/puc/blob/HEAD/prompts/loader.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"45c510e0b8adafbc","mcp_get_code":{"code_sha256":"45c510e0b8adafbc"}},{"arxiv_id":"2609.01888","paper":"/paper/arxiv-2609-01888","title":"Does Playing it Safe Count as Faithfulness? Reassessing LVLM Hallucination Mitigation Methods","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"mehrdadfazli/AssessHalVLM","path":"methods/after/utils.py","file_url":"https://github.com/mehrdadfazli/AssessHalVLM/blob/HEAD/methods/after/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0d53582061af2e49","mcp_get_code":{"code_sha256":"0d53582061af2e49"}},{"arxiv_id":"2607.27379","paper":"/paper/arxiv-2607-27379","title":"HSS-Synth: Humanities and Social Sciences Data Synthesis for LLMs","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"pengr/HSS-Synth","path":"code/baseline/syn_method/LongForm.py","file_url":"https://github.com/pengr/HSS-Synth/blob/HEAD/code/baseline/syn_method/LongForm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"53c6e2a7675a4a40","mcp_get_code":{"code_sha256":"53c6e2a7675a4a40"}},{"arxiv_id":"2606.15307","paper":"/paper/arxiv-2606-15307","title":"Adapting Reinforcement Learning with Chain-of-Thought Supervision for Explainable Detection of Hateful and Propagandistic Memes","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"MohamedBayan/MemeReason","path":"data_prep/prepare_training_data.py","file_url":"https://github.com/MohamedBayan/MemeReason/blob/HEAD/data_prep/prepare_training_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f291f6bb0f74ccc4","mcp_get_code":{"code_sha256":"f291f6bb0f74ccc4"}},{"arxiv_id":"2606.02800","paper":"/paper/arxiv-2606-02800","title":"Cosmos 3: Omnimodal World Models for Physical AI NVIDIA","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"nvidia/cosmos","path":"evaluation/cosmos3/generator/unigenbench/ugb_scorer.py","file_url":"https://github.com/nvidia/cosmos/blob/HEAD/evaluation/cosmos3/generator/unigenbench/ugb_scorer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"231096399f905d44","mcp_get_code":{"code_sha256":"231096399f905d44"}},{"arxiv_id":"2605.13412","paper":"/paper/arxiv-2605-13412","title":"LLMs as annotators of credibility assessment in Danish asylum decisions: evaluating classification performance and errors beyond aggregated metrics","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"glhr/RAB-Cred","path":"llm_annotation/llm_utils.py","file_url":"https://github.com/glhr/RAB-Cred/blob/HEAD/llm_annotation/llm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8e9614852bb8afd7","mcp_get_code":{"code_sha256":"8e9614852bb8afd7"}},{"arxiv_id":"2605.08083","paper":"/paper/arxiv-2605-08083","title":"LLMs Improving LLMs: Agentic Discovery for Test-Time Scaling","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"zhengkid/AutoTTS","path":"efficient_reasoning_controller/workspace/controller_search/claude_proposer.py","file_url":"https://github.com/zhengkid/AutoTTS/blob/HEAD/efficient_reasoning_controller/workspace/controller_search/claude_proposer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c66918620c0773fb","mcp_get_code":{"code_sha256":"c66918620c0773fb"}},{"arxiv_id":"2604.19459","paper":"/paper/arxiv-2604-19459","title":"Do LLMs Game Formalization? Evaluating Faithfulness in Logical Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"koreankiwi99/formalization-gaming","path":"src/utils/prompts.py","file_url":"https://github.com/koreankiwi99/formalization-gaming/blob/HEAD/src/utils/prompts.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"872d7c12ef1aca3f","mcp_get_code":{"code_sha256":"872d7c12ef1aca3f"}},{"arxiv_id":"2603.26469","paper":"/paper/arxiv-2603-26469","title":"UNIFERENCE: A Discrete Event Simulation Framework for Developing Distributed AI Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Dogacel/Uniference","path":"benchmarks/commons.py","file_url":"https://github.com/Dogacel/Uniference/blob/HEAD/benchmarks/commons.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"f2c1a078e541b975","mcp_get_code":{"code_sha256":"f2c1a078e541b975"}},{"arxiv_id":"2602.11509","paper":"/paper/arxiv-2602-11509","title":"Multimodal Fact-Level Attribution for Verifiable Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"meetdavidwan/murgat","path":"src/program_utils.py","file_url":"https://github.com/meetdavidwan/murgat/blob/HEAD/src/program_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f8d44f2e394301ff","mcp_get_code":{"code_sha256":"f8d44f2e394301ff"}},{"arxiv_id":"2601.08955","paper":"/paper/arxiv-2601-08955","title":"Imagine-then-Plan: Agent Learning from Adaptive Lookahead with World Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"loyiv/ITP","path":"itp/orchestrator.py","file_url":"https://github.com/loyiv/ITP/blob/HEAD/itp/orchestrator.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"21523634b7dd82f6","mcp_get_code":{"code_sha256":"21523634b7dd82f6"}},{"arxiv_id":"2601.05654","paper":"/paper/arxiv-2601-05654","title":"Learning to Retrieve User History and Generate User Profiles for Personalized Persuasiveness Prediction","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"holi-lab/ReCAP","path":"recap/profiler/profiler.py","file_url":"https://github.com/holi-lab/ReCAP/blob/HEAD/recap/profiler/profiler.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"40709e284e08c1e1","mcp_get_code":{"code_sha256":"40709e284e08c1e1"}},{"arxiv_id":"2510.20342","paper":"/paper/arxiv-2510-20342","title":"Teaching Language Models to Reason with Tools","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"ChengpengLi1003/CoRT","path":"infer/utils.py","file_url":"https://github.com/ChengpengLi1003/CoRT/blob/HEAD/infer/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3bb02b5083de68fd","mcp_get_code":{"code_sha256":"3bb02b5083de68fd"}},{"arxiv_id":"2503.05188","paper":null,"title":"arXiv:2503.05188","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":null,"repo":"BugMakerzzz/CRISP","path":"crisp_reason.py","file_url":"https://github.com/BugMakerzzz/CRISP/blob/HEAD/crisp_reason.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"caa4cec487be6f06","mcp_get_code":{"code_sha256":"caa4cec487be6f06"}},{"arxiv_id":"2412.15529","paper":"/paper/xrag-examining-the-core-benchmarking","title":"XRAG: eXamining the Core -- Benchmarking Foundational Components in Advanced Retrieval-Augmented Generation","date":"2024-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"docailab/xrag","path":"src/xrag/adaptive_rag/utils.py","file_url":"https://github.com/docailab/xrag/blob/HEAD/src/xrag/adaptive_rag/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"71f8c85c5d4c825c","mcp_get_code":{"code_sha256":"71f8c85c5d4c825c"}},{"arxiv_id":"2411.09502","paper":"/paper/golden-noise-for-diffusion-models-a-learning","title":"Golden Noise for Diffusion Models: A Learning Framework","date":"2024-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xie-lab-ml/golden-noise-for-diffusion-models","path":"training/dataset_collection.py","file_url":"https://github.com/xie-lab-ml/golden-noise-for-diffusion-models/blob/HEAD/training/dataset_collection.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"94040482358d6999","mcp_get_code":{"code_sha256":"94040482358d6999"}},{"arxiv_id":"2407.18064","paper":"/paper/compeer-a-generative-conversational-agent-for","title":"ComPeer: A Generative Conversational Agent for Proactive Peer Support","date":null,"month_inferred_from_arxiv_id":"2024-07","title_source":"archive","repo":"liutj9/compeer","path":"src/tools.py","file_url":"https://github.com/liutj9/compeer/blob/HEAD/src/tools.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f3f8f7c9479f15e7","mcp_get_code":{"code_sha256":"f3f8f7c9479f15e7"}},{"arxiv_id":"2402.16048","paper":"/paper/llms-with-chain-of-thought-are-non-causal","title":"How Likely Do LLMs with CoT Mimic Human Reasoning?","date":"2024-02-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stevenzhb/cot_causal_analysis","path":"scripts/api_run.py","file_url":"https://github.com/stevenzhb/cot_causal_analysis/blob/HEAD/scripts/api_run.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7c56d76aa689e7f0","mcp_get_code":{"code_sha256":"7c56d76aa689e7f0"}},{"arxiv_id":"2402.11900","paper":"/paper/investigating-multi-hop-factual-shortcuts-in","title":"Investigating Multi-Hop Factual Shortcuts in Knowledge Editing of Large Language Models","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Jometeorie/MultiHopShortcuts","path":"code/data_utils.py","file_url":"https://github.com/Jometeorie/MultiHopShortcuts/blob/HEAD/code/data_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"f1273794d66182e4","mcp_get_code":{"code_sha256":"f1273794d66182e4"}},{"arxiv_id":"2312.17115","paper":"/paper/how-far-are-we-from-believable-ai-agents-a","title":"How Far Are LLMs from Believable AI? A Benchmark for Evaluating the Believability of Human Behavior Simulation","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llmconference/emnlp_conference_2024","path":"benchmark/benchmark_file_util.py","file_url":"https://github.com/llmconference/emnlp_conference_2024/blob/HEAD/benchmark/benchmark_file_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"099db8c2586d17b9","mcp_get_code":{"code_sha256":"099db8c2586d17b9"}},{"arxiv_id":"2309.17452","paper":"/paper/tora-a-tool-integrated-reasoning-agent-for","title":"ToRA: A Tool-Integrated Reasoning Agent for Mathematical Problem Solving","date":"2023-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/tora","path":"src/utils/utils.py","file_url":"https://github.com/microsoft/tora/blob/HEAD/src/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3bb02b5083de68fd","mcp_get_code":{"code_sha256":"3bb02b5083de68fd"}},{"arxiv_id":"2305.20076","paper":"/paper/decision-oriented-dialogue-for-human-ai","title":"Decision-Oriented Dialogue for Human-AI Collaboration","date":"2023-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jlin816/dialop","path":"dialop/play_gpt3.py","file_url":"https://github.com/jlin816/dialop/blob/HEAD/dialop/play_gpt3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"7e94d97091b55364","mcp_get_code":{"code_sha256":"7e94d97091b55364"}},{"arxiv_id":"2305.11738","paper":"/paper/critic-large-language-models-can-self-correct","title":"CRITIC: Large Language Models Can Self-Correct with Tool-Interactive Critiquing","date":"2023-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/ProphetNet","path":"CRITIC/src/program/critic.py","file_url":"https://github.com/microsoft/ProphetNet/blob/HEAD/CRITIC/src/program/critic.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d9d6df1c6dfcbd38","mcp_get_code":{"code_sha256":"d9d6df1c6dfcbd38"}},{"arxiv_id":"2210.03629","paper":"/paper/react-synergizing-reasoning-and-acting-in","title":"ReAct: Synergizing Reasoning and Acting in Language Models","date":"2022-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liyuan24/nanodeepresearch","path":"agent/react_agent.py","file_url":"https://github.com/liyuan24/nanodeepresearch/blob/HEAD/agent/react_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb56e34b15e5e739","mcp_get_code":{"code_sha256":"cb56e34b15e5e739"}},{"arxiv_id":"2208.11057","paper":"/paper/prompting-as-probing-using-language-models","title":"Prompting as Probing: Using Language Models for Knowledge Base Construction","date":"2022-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hemile/iswc-challenge","path":"gpt3_baseline.py","file_url":"https://github.com/hemile/iswc-challenge/blob/HEAD/gpt3_baseline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bc5c5e8713300ce5","mcp_get_code":{"code_sha256":"bc5c5e8713300ce5"}}]}