{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-question","entry":"get_question","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":12,"n_papers_ran":3,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":3,"n_samples_fingerprinted":0,"n_places":12,"n_places_pointer_only":6,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":2,"unverified":6},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.07510","paper":"/paper/arxiv-2605-07510","title":"InterLV-Search: Benchmarking Interleaved Multimodal Agentic Search","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"hbhalpha/InterLV-Search-Bench","path":"agentic_search/gpt_judge.py","file_url":"https://github.com/hbhalpha/InterLV-Search-Bench/blob/HEAD/agentic_search/gpt_judge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9aec6bbb5eec953f","mcp_get_code":{"code_sha256":"9aec6bbb5eec953f"}},{"arxiv_id":"2602.01685","paper":"/paper/arxiv-2602-01685","title":"Semantic-aware Wasserstein Policy Regularization for Large Language Model Alignment","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"aailab-kaist/WPR","path":"inference/testing_util.py","file_url":"https://github.com/aailab-kaist/WPR/blob/HEAD/inference/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ed8331339b4de97c","mcp_get_code":{"code_sha256":"ed8331339b4de97c"}},{"arxiv_id":"2501.04694","paper":"/paper/epicoder-encompassing-diversity-and","title":"EpiCoder: Encompassing Diversity and Complexity in Code Generation","date":"2025-01-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DeepLearnXMU/EpiCoder","path":"gen/gen_code.py","file_url":"https://github.com/DeepLearnXMU/EpiCoder/blob/HEAD/gen/gen_code.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"91751933774cda1d","mcp_get_code":{"code_sha256":"91751933774cda1d"}},{"arxiv_id":"2406.06462","paper":"/paper/vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianyu-z/VCR","path":"src/evaluation/closed_source_eval.py","file_url":"https://github.com/tianyu-z/VCR/blob/HEAD/src/evaluation/closed_source_eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"CC-BY-SA-4.0","inline_ok":false,"code_sha256_prefix":"dcb7f3e37ca7aaef","mcp_get_code":{"code_sha256":"dcb7f3e37ca7aaef"}},{"arxiv_id":"2403.04706","paper":"/paper/common-7b-language-models-already-possess","title":"Common 7B Language Models Already Possess Strong Math Capabilities","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xwin-lm/xwin-lm","path":"Xwin-Coder/APPS/testing_util.py","file_url":"https://github.com/xwin-lm/xwin-lm/blob/HEAD/Xwin-Coder/APPS/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ed8331339b4de97c","mcp_get_code":{"code_sha256":"ed8331339b4de97c"}},{"arxiv_id":"2402.16893","paper":"/paper/the-good-and-the-bad-exploring-privacy-issues","title":"The Good and The Bad: Exploring Privacy Issues in Retrieval-Augmented Generation (RAG)","date":"2024-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"phycholosogy/rag-privacy","path":"generate_prompt.py","file_url":"https://github.com/phycholosogy/rag-privacy/blob/HEAD/generate_prompt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ec79ef678f8b2a96","mcp_get_code":{"code_sha256":"ec79ef678f8b2a96"}},{"arxiv_id":"2310.12836","paper":"/paper/knowledge-augmented-language-model","title":"Knowledge-Augmented Language Model Verification","date":"2023-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jinheonbaek/kalmv","path":"preprocess/process_odqa.py","file_url":"https://github.com/jinheonbaek/kalmv/blob/HEAD/preprocess/process_odqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a6f87877056e2eb3","mcp_get_code":{"code_sha256":"a6f87877056e2eb3"}},{"arxiv_id":"2203.08597","paper":"/paper/less-is-more-summary-of-long-instructions-is","title":"Less is More: Summary of Long Instructions is Better for Program Synthesis","date":"2022-03-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kurbster/prompt-summarization","path":"src/lib/testing_util.py","file_url":"https://github.com/kurbster/prompt-summarization/blob/HEAD/src/lib/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f452e011dda0ca8f","mcp_get_code":{"code_sha256":"f452e011dda0ca8f"}},{"arxiv_id":"2107.03374","paper":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codedotal/gpt-code-clippy","path":"evaluation/apps_eval_util.py","file_url":"https://github.com/codedotal/gpt-code-clippy/blob/HEAD/evaluation/apps_eval_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ed8331339b4de97c","mcp_get_code":{"code_sha256":"ed8331339b4de97c"}},{"arxiv_id":"2105.09938","paper":"/paper/measuring-coding-challenge-competence-with","title":"Measuring Coding Challenge Competence With APPS","date":"2021-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hendrycks/apps","path":"eval/testing_util.py","file_url":"https://github.com/hendrycks/apps/blob/HEAD/eval/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ed8331339b4de97c","mcp_get_code":{"code_sha256":"ed8331339b4de97c"}},{"arxiv_id":"2025.findings-acl.444","paper":null,"title":"arXiv:2025.findings-acl.444","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Gebro13/PRM_research","path":"Data_Preprocess/sentenceGET.py","file_url":"https://github.com/Gebro13/PRM_research/blob/HEAD/Data_Preprocess/sentenceGET.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4652ecf30326f298","mcp_get_code":{"code_sha256":"4652ecf30326f298"}},{"arxiv_id":"2024.findings-emnlp.253","paper":null,"title":"arXiv:2024.findings-emnlp.253","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WilliamZR/ProTrix","path":"evaluation/eval_with_llm.py","file_url":"https://github.com/WilliamZR/ProTrix/blob/HEAD/evaluation/eval_with_llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5791edbc3c4dc955","mcp_get_code":{"code_sha256":"5791edbc3c4dc955"}}]}