{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-answer","entry":"get_answer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":24,"n_papers_ran":15,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":23,"n_samples_ran":14,"n_samples_fingerprinted":4,"n_places":25,"n_places_pointer_only":15,"by_status":{"ran_honours":0,"ran_violates":1,"ran_draft_wrong":4,"ran_fixture":0,"ran":9,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2603.02938","paper":"/paper/arxiv-2603-02938","title":"Beyond One-Size-Fits-All: Adaptive Subgraph Denoising for Zero-Shot Graph Learning with Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"mysteriouslfz/GraphSSR","path":"evaluation/evaluate_trained_model.py","file_url":"https://github.com/mysteriouslfz/GraphSSR/blob/HEAD/evaluation/evaluate_trained_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9f5c3b50b2260372","mcp_get_code":{"code_sha256":"9f5c3b50b2260372"}},{"arxiv_id":"2601.14044","paper":"/paper/arxiv-2601-14044","title":"Weather-R1: Logically Consistent Reinforcement Fine-Tuning for Multimodal Reasoning in Meteorology","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Marcowky/Weather-R1","path":"src/weather_r1/weather_r1_reward.py","file_url":"https://github.com/Marcowky/Weather-R1/blob/HEAD/src/weather_r1/weather_r1_reward.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MPL-2.0","inline_ok":false,"code_sha256_prefix":"425ad6851442a3c2","mcp_get_code":{"code_sha256":"425ad6851442a3c2"}},{"arxiv_id":"2511.13223","paper":"/paper/arxiv-2511-13223","title":"TokenSqueeze: Performance-Preserving Compression for Reasoning LLMs","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"zhangyx1122/TokenSqueeze","path":"utils/tools.py","file_url":"https://github.com/zhangyx1122/TokenSqueeze/blob/HEAD/utils/tools.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4eed1f4a7be28b06","mcp_get_code":{"code_sha256":"4eed1f4a7be28b06"}},{"arxiv_id":"2502.15828","paper":"/paper/a-stronger-mixture-of-low-rank-experts-for","title":"A Stronger Mixture of Low-Rank Experts for Fine-Tuning Foundation Models","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/moelora_riemannian","path":"prepare_data.py","file_url":"https://github.com/thudm/moelora_riemannian/blob/HEAD/prepare_data.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"30a82d453ae174bf","mcp_get_code":{"code_sha256":"30a82d453ae174bf"}},{"arxiv_id":"2501.12835","paper":"/paper/adaptive-retrieval-without-self-knowledge","title":"Adaptive Retrieval Without Self-Knowledge? Bringing Uncertainty Back Home","date":"2025-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"s-nlp/AdaRAGUE","path":"SeaKR/main_simpleqa.py","file_url":"https://github.com/s-nlp/AdaRAGUE/blob/HEAD/SeaKR/main_simpleqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"64fc92f1ad0464da","mcp_get_code":{"code_sha256":"64fc92f1ad0464da"}},{"arxiv_id":"2412.19513","paper":"/paper/confidence-v-s-critique-a-decomposition-of","title":"Confidence v.s. Critique: A Decomposition of Self-Correction Capability for LLMs","date":"2024-12-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Zhe-Young/SelfCorrectDecompose","path":"eval_logits.py","file_url":"https://github.com/Zhe-Young/SelfCorrectDecompose/blob/HEAD/eval_logits.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e888e7e76d03f6fa","mcp_get_code":{"code_sha256":"e888e7e76d03f6fa"}},{"arxiv_id":"2407.10223","paper":"/paper/practical-unlearning-for-large-language","title":"On Large Language Model Continual Unlearning","date":"2024-07-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GCYZSL/O3-LLM-UNLEARNING","path":"ScienceQA_expriment/preprocess_scienceqa.py","file_url":"https://github.com/GCYZSL/O3-LLM-UNLEARNING/blob/HEAD/ScienceQA_expriment/preprocess_scienceqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2579b1304396d99b","mcp_get_code":{"code_sha256":"2579b1304396d99b"}},{"arxiv_id":"2407.04363","paper":"/paper/arigraph-learning-knowledge-graph-world","title":"AriGraph: Learning Knowledge Graph World Models with Episodic Memory for LLM Agents","date":"2024-07-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"airi-institute/arigraph","path":"musique_test_big.py","file_url":"https://github.com/airi-institute/arigraph/blob/HEAD/musique_test_big.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d7983fb0733e663e","mcp_get_code":{"code_sha256":"d7983fb0733e663e"}},{"arxiv_id":"2406.19215","paper":"/paper/seakr-self-aware-knowledge-retrieval-for","title":"SeaKR: Self-aware Knowledge Retrieval for Adaptive Retrieval Augmented Generation","date":"2024-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-keg/seakr","path":"main_simpleqa.py","file_url":"https://github.com/thu-keg/seakr/blob/HEAD/main_simpleqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"64fc92f1ad0464da","mcp_get_code":{"code_sha256":"64fc92f1ad0464da"}},{"arxiv_id":"2406.05654","paper":"/paper/domainrag-a-chinese-benchmark-for-evaluating","title":"DomainRAG: A Chinese Benchmark for Evaluating Domain-specific Retrieval-Augmented Generation","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShootingWong/DomainRAG","path":"BCM/models/llm_models.py","file_url":"https://github.com/ShootingWong/DomainRAG/blob/HEAD/BCM/models/llm_models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0569ac90f9042b0d","mcp_get_code":{"code_sha256":"0569ac90f9042b0d"}},{"arxiv_id":"2406.05654","paper":"/paper/domainrag-a-chinese-benchmark-for-evaluating","title":"DomainRAG: A Chinese Benchmark for Evaluating Domain-specific Retrieval-Augmented Generation","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ShootingWong/DomainRAG","path":"BCM/evaluation/prediction_results/gpt_eval.py","file_url":"https://github.com/ShootingWong/DomainRAG/blob/HEAD/BCM/evaluation/prediction_results/gpt_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ad0fc83e3ddfae96","mcp_get_code":{"code_sha256":"ad0fc83e3ddfae96"}},{"arxiv_id":"2404.08589","paper":"/paper/enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ovguyo/captions-in-vqa","path":"qa.py","file_url":"https://github.com/ovguyo/captions-in-vqa/blob/HEAD/qa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32aeeae5aebd3548","mcp_get_code":{"code_sha256":"32aeeae5aebd3548"}},{"arxiv_id":"2403.05004","paper":"/paper/can-t-remember-details-in-long-documents-you","title":"Can't Remember Details in Long Documents? You Need Some R&R","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"casetext/r-and-r","path":"src/prompts.py","file_url":"https://github.com/casetext/r-and-r/blob/HEAD/src/prompts.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1aa6ab3d47a5bf82","mcp_get_code":{"code_sha256":"1aa6ab3d47a5bf82"}},{"arxiv_id":"2403.00425","paper":"/paper/halc-object-hallucination-reduction-via","title":"HALC: Object Hallucination Reduction via Adaptive Focal-Contrast Decoding","date":"2024-03-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BillChan226/HALC","path":"models/answerer.py","file_url":"https://github.com/BillChan226/HALC/blob/HEAD/models/answerer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a0c217c9a4b3e68c","mcp_get_code":{"code_sha256":"a0c217c9a4b3e68c"}},{"arxiv_id":"2402.08562","paper":"/paper/higher-layers-need-more-lora-experts","title":"Higher Layers Need More LoRA Experts","date":"2024-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gcyzsl/mola","path":"preparation_scienceqa_data.py","file_url":"https://github.com/gcyzsl/mola/blob/HEAD/preparation_scienceqa_data.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"30a82d453ae174bf","mcp_get_code":{"code_sha256":"30a82d453ae174bf"}},{"arxiv_id":"2401.05618","paper":"/paper/the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"matthewrenze/jhu-concise-cot","path":"source/actions.py","file_url":"https://github.com/matthewrenze/jhu-concise-cot/blob/HEAD/source/actions.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"5784148c3597808f","mcp_get_code":{"code_sha256":"5784148c3597808f"}},{"arxiv_id":"2310.16045","paper":"/paper/woodpecker-hallucination-correction-for","title":"Woodpecker: Hallucination Correction for Multimodal Large Language Models","date":"2023-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bradyfu/woodpecker","path":"models/answerer.py","file_url":"https://github.com/bradyfu/woodpecker/blob/HEAD/models/answerer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"20117a653c8f40ac","mcp_get_code":{"code_sha256":"20117a653c8f40ac"}},{"arxiv_id":"2310.12836","paper":"/paper/knowledge-augmented-language-model","title":"Knowledge-Augmented Language Model Verification","date":"2023-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jinheonbaek/kalmv","path":"preprocess/process_odqa.py","file_url":"https://github.com/jinheonbaek/kalmv/blob/HEAD/preprocess/process_odqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2472555bb4ad6b7e","mcp_get_code":{"code_sha256":"2472555bb4ad6b7e"}},{"arxiv_id":"2309.08594","paper":"/paper/merge-conflicts-exploring-the-impacts-of","title":"\"Merge Conflicts!\" Exploring the Impacts of External Distractors to Parametric Knowledge Graphs","date":"2023-09-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qiancheng0/ekd_impacts_pkg","path":"utils_graph.py","file_url":"https://github.com/qiancheng0/ekd_impacts_pkg/blob/HEAD/utils_graph.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dc30d94315c62317","mcp_get_code":{"code_sha256":"dc30d94315c62317"}},{"arxiv_id":"2304.09667","paper":"/paper/genegpt-teaching-large-language-models-to-use","title":"GeneGPT: Augmenting Large Language Models with Domain Tools for Improved Access to Biomedical Information","date":"2023-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ncbi/GeneGPT","path":"evaluate.py","file_url":"https://github.com/ncbi/GeneGPT/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"4b2ca03bb76ec33c","mcp_get_code":{"code_sha256":"4b2ca03bb76ec33c"}},{"arxiv_id":"1805.10196","paper":"/paper/maximizing-acquisition-functions-for-bayesian","title":"Maximizing acquisition functions for Bayesian optimization","date":"2018-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"j-wilson/MaximizingAcquisitionFunctions","path":"scripts/run_bayesopt.py","file_url":"https://github.com/j-wilson/MaximizingAcquisitionFunctions/blob/HEAD/scripts/run_bayesopt.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0ffba6a9e9a282fa","mcp_get_code":{"code_sha256":"0ffba6a9e9a282fa"}},{"arxiv_id":"1706.04223","paper":"/paper/adversarially-regularized-autoencoders","title":"Adversarially Regularized Autoencoders","date":"2017-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lingofunk/lingofunk-transfer-style","path":"lingofunk_transfer_style/__main__.py","file_url":"https://github.com/lingofunk/lingofunk-transfer-style/blob/HEAD/lingofunk_transfer_style/__main__.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"1a66e1c4bac566fd","mcp_get_code":{"code_sha256":"1a66e1c4bac566fd"}},{"arxiv_id":"2025.acl-long.806","paper":null,"title":"arXiv:2025.acl-long.806","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"lzw108/RAEmoLLM","path":"retrieval/retrieval_COCO.py","file_url":"https://github.com/lzw108/RAEmoLLM/blob/HEAD/retrieval/retrieval_COCO.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8f4d60342a4c7114","mcp_get_code":{"code_sha256":"8f4d60342a4c7114"}},{"arxiv_id":"2024.findings-emnlp.253","paper":null,"title":"arXiv:2024.findings-emnlp.253","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WilliamZR/ProTrix","path":"evaluation/eval_with_llm.py","file_url":"https://github.com/WilliamZR/ProTrix/blob/HEAD/evaluation/eval_with_llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"500a45eacde0ca09","mcp_get_code":{"code_sha256":"500a45eacde0ca09"}},{"arxiv_id":"2023.findings-emnlp.1023","paper":null,"title":"arXiv:2023.findings-emnlp.1023","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HKUST-KnowComp/QaDynamics","path":"src/Data_generation/filter_CWWV.py","file_url":"https://github.com/HKUST-KnowComp/QaDynamics/blob/HEAD/src/Data_generation/filter_CWWV.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"47e378f51ba3827d","mcp_get_code":{"code_sha256":"47e378f51ba3827d"}}]}