{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/process-question","entry":"process_question","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":10,"n_papers_ran":5,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":5,"n_samples_fingerprinted":1,"n_places":11,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":2,"unverified":6},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.21867","paper":"/paper/arxiv-2606-21867","title":"ForEx: A Formal Verification Framework for Explainable Reasoning in Logical Fallacy Detection and Annotation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"D3-Laboratory/ForEx","path":"src/experiment_processor.py","file_url":"https://github.com/D3-Laboratory/ForEx/blob/HEAD/src/experiment_processor.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ad5b86651ffa4d6f","mcp_get_code":{"code_sha256":"ad5b86651ffa4d6f"}},{"arxiv_id":"2604.27453","paper":"/paper/arxiv-2604-27453","title":"From Coarse to Fine: Benchmarking and Reward Modeling for Writing-Centric Generation Tasks","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Rainier-rq1/From_Coarse_to_Fine","path":"WEval/WritingBench-Critic.py","file_url":"https://github.com/Rainier-rq1/From_Coarse_to_Fine/blob/HEAD/WEval/WritingBench-Critic.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"35823a35719c0bdf","mcp_get_code":{"code_sha256":"35823a35719c0bdf"}},{"arxiv_id":"2604.08281","paper":"/paper/arxiv-2604-08281","title":"When to Trust Tools? Adaptive Tool Trust Calibration For Tool-Integrated Math Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"00Dreamer00/ATTC","path":"add_new_dataset.py","file_url":"https://github.com/00Dreamer00/ATTC/blob/HEAD/add_new_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"48fad3d8f7e025e9","mcp_get_code":{"code_sha256":"48fad3d8f7e025e9"}},{"arxiv_id":"2601.11047","paper":"/paper/arxiv-2601-11047","title":"CoG: Controllable Graph Reasoning via Relational Blueprints and Failure-Aware Refinement over Knowledge Graphs","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"zjukg/CoG","path":"CoG/main_freebase.py","file_url":"https://github.com/zjukg/CoG/blob/HEAD/CoG/main_freebase.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d1f04ebb317abd0e","mcp_get_code":{"code_sha256":"d1f04ebb317abd0e"}},{"arxiv_id":"2504.20965","paper":"/paper/aegisllm-scaling-agentic-systems-for-self","title":"AegisLLM: Scaling Agentic Systems for Self-Reflective Defense in LLM Security","date":"2025-04-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zikuicai/aegisllm","path":"run_unlearning_mcq.py","file_url":"https://github.com/zikuicai/aegisllm/blob/HEAD/run_unlearning_mcq.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f362b0a173e63a2d","mcp_get_code":{"code_sha256":"f362b0a173e63a2d"}},{"arxiv_id":"2503.08688","paper":"/paper/randomness-not-representation-the","title":"Randomness, Not Representation: The Unreliability of Evaluating Cultural Alignment in LLMs","date":null,"month_inferred_from_arxiv_id":"2025-03","title_source":"archive","repo":"ariba-k/llm-cultural-alignment-evaluation","path":"steerability/run_steerability.py","file_url":"https://github.com/ariba-k/llm-cultural-alignment-evaluation/blob/HEAD/steerability/run_steerability.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f8b212f56c3563ac","mcp_get_code":{"code_sha256":"f8b212f56c3563ac"}},{"arxiv_id":"2410.18417","paper":"/paper/large-language-models-reflect-the-ideology-of","title":"Large Language Models Reflect the Ideology of their Creators","date":"2024-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aida-ugent/llm-ideology-analysis","path":"src/run_questions_through_unified_api.py","file_url":"https://github.com/aida-ugent/llm-ideology-analysis/blob/HEAD/src/run_questions_through_unified_api.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"efde6c9a51f02cd6","mcp_get_code":{"code_sha256":"efde6c9a51f02cd6"}},{"arxiv_id":"2312.17115","paper":"/paper/how-far-are-we-from-believable-ai-agents-a","title":"How Far Are LLMs from Believable AI? A Benchmark for Evaluating the Believability of Human Behavior Simulation","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llmconference/emnlp_conference_2024","path":"benchmark/benchmark_class.py","file_url":"https://github.com/llmconference/emnlp_conference_2024/blob/HEAD/benchmark/benchmark_class.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dc5b4a64081c2586","mcp_get_code":{"code_sha256":"dc5b4a64081c2586"}},{"arxiv_id":"2207.00383","paper":"/paper/reler-zju-alibaba-submission-to-the-ego4d","title":"ReLER@ZJU-Alibaba Submission to the Ego4D Natural Language Queries Challenge 2022","date":"2022-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nnnnai/ego4d_nlq_2022_1st_place_solution","path":"ms_cm/vslnet_utils/prepare_ego4d_dataset.py","file_url":"https://github.com/nnnnai/ego4d_nlq_2022_1st_place_solution/blob/HEAD/ms_cm/vslnet_utils/prepare_ego4d_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"eff6500b503388a1","mcp_get_code":{"code_sha256":"eff6500b503388a1"}},{"arxiv_id":"1803.03067","paper":"/paper/compositional-attention-networks-for-machine","title":"Compositional Attention Networks for Machine Reasoning","date":"2018-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ceyzaguirre4/mac-network-pytorch","path":"preprocess.py","file_url":"https://github.com/ceyzaguirre4/mac-network-pytorch/blob/HEAD/preprocess.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"928267eac64ad657","mcp_get_code":{"code_sha256":"928267eac64ad657"}},{"arxiv_id":"1803.03067","paper":"/paper/compositional-attention-networks-for-machine","title":"Compositional Attention Networks for Machine Reasoning","date":"2018-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rosinality/mac-network-pytorch","path":"preprocess.py","file_url":"https://github.com/rosinality/mac-network-pytorch/blob/HEAD/preprocess.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6490066d39c2eafd","mcp_get_code":{"code_sha256":"6490066d39c2eafd"}}]}