{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/pass-at-k","entry":"pass_at_k","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":16,"n_papers_ran":12,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":12,"n_samples_ran":8,"n_samples_fingerprinted":8,"n_places":16,"n_places_pointer_only":6,"by_status":{"ran_honours":7,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.29632","paper":"/paper/arxiv-2608-29632","title":"InteractBench: Benchmarking LLMs on Competitive Programming under Unrevealed Information","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"kmsgk0/InteractBench","path":"evaluate.py","file_url":"https://github.com/kmsgk0/InteractBench/blob/HEAD/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a28acf9b2f202312","mcp_get_code":{"code_sha256":"a28acf9b2f202312"}},{"arxiv_id":"2605.17937","paper":"/paper/arxiv-2605-17937","title":"BacktestBench: Benchmarking Large Language Models for Automated Quantitative Strategy Backtesting","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"jensenw1/BacktestBench","path":"AutoBacktest/002_SQL/utils.py","file_url":"https://github.com/jensenw1/BacktestBench/blob/HEAD/AutoBacktest/002_SQL/utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"62af43e66bbbc806","mcp_get_code":{"code_sha256":"62af43e66bbbc806"}},{"arxiv_id":"2604.14877","paper":"/paper/arxiv-2604-14877","title":"Does RL Expand the Capability Boundary of LLM Agents? A PASS@(k, T ) Analysis *","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zhiyuanZhai20/pass-kt-analysis","path":"eval/compute_pass_kt.py","file_url":"https://github.com/zhiyuanZhai20/pass-kt-analysis/blob/HEAD/eval/compute_pass_kt.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"340437ebc52f7a3a","mcp_get_code":{"code_sha256":"340437ebc52f7a3a"}},{"arxiv_id":"2602.19517","paper":"/paper/arxiv-2602-19517","title":"Classroom Final Exam: An Instructor-Tested Reasoning Benchmark","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Analogy-AI/CFE_Bench","path":"evaluation.py","file_url":"https://github.com/Analogy-AI/CFE_Bench/blob/HEAD/evaluation.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7bbf842f40dbc128","mcp_get_code":{"code_sha256":"7bbf842f40dbc128"}},{"arxiv_id":"2602.11639","paper":"/paper/arxiv-2602-11639","title":"PACE: Prefix-Protected and Difficulty-Aware Compression for Efficient Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"LiveCodeBench/LiveCodeBench","path":"lcb_runner/evaluation/compute_code_execution_metrics.py","file_url":"https://github.com/LiveCodeBench/LiveCodeBench/blob/HEAD/lcb_runner/evaluation/compute_code_execution_metrics.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3a7c7ec16769928a","mcp_get_code":{"code_sha256":"3a7c7ec16769928a"}},{"arxiv_id":"2507.13266","paper":"/paper/questa-expanding-reasoning-capacity-in-llms","title":"QuestA: Expanding Reasoning Capacity in LLMs via Question Augmentation","date":"2025-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"foreverlasting1202/QuestA","path":"AReaL/evaluation/eval_and_aggregate.py","file_url":"https://github.com/foreverlasting1202/QuestA/blob/HEAD/AReaL/evaluation/eval_and_aggregate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5a4b219d1513f202","mcp_get_code":{"code_sha256":"5a4b219d1513f202"}},{"arxiv_id":"2506.12713","paper":"/paper/humanity-s-last-code-exam-can-advanced-llms","title":"Humanity's Last Code Exam: Can Advanced LLMs Conquer Human's Hardest Code Competition?","date":"2025-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"3a7c7ec16769928a","mcp_get_code":{"code_sha256":"3a7c7ec16769928a"}},{"arxiv_id":"2505.18384","paper":"/paper/dynamic-risk-assessments-for-offensive","title":"Dynamic Risk Assessments for Offensive Cybersecurity Agents","date":"2025-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"boyiwei/dynamic-risk-assessment","path":"analysis/grade_benchmark.py","file_url":"https://github.com/boyiwei/dynamic-risk-assessment/blob/HEAD/analysis/grade_benchmark.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"31dac9f8d6c131b8","mcp_get_code":{"code_sha256":"31dac9f8d6c131b8"}},{"arxiv_id":"2504.11456","paper":"/paper/deepmath-103k-a-large-scale-challenging","title":"DeepMath-103K: A Large-Scale, Challenging, Decontaminated, and Verifiable Mathematical Dataset for Advancing Reasoning","date":"2025-04-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwhe99/deepmath","path":"uni_eval.py","file_url":"https://github.com/zwhe99/deepmath/blob/HEAD/uni_eval.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e3014276a256dd64","mcp_get_code":{"code_sha256":"e3014276a256dd64"}},{"arxiv_id":"2503.05592","paper":"/paper/r1-searcher-incentivizing-the-search","title":"R1-Searcher: Incentivizing the Search Capability in LLMs via Reinforcement Learning","date":"2025-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/simpledeepsearcher","path":"inference/lcb_runner/evaluation/compute_code_execution_metrics.py","file_url":"https://github.com/rucaibox/simpledeepsearcher/blob/HEAD/inference/lcb_runner/evaluation/compute_code_execution_metrics.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3a7c7ec16769928a","mcp_get_code":{"code_sha256":"3a7c7ec16769928a"}},{"arxiv_id":"2410.22480","paper":"/paper/scaling-llm-inference-with-optimized-sample","title":"Scaling LLM Inference with Optimized Sample Compute Allocation","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leililab/osca","path":"osca.py","file_url":"https://github.com/leililab/osca/blob/HEAD/osca.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3f3c074604757720","mcp_get_code":{"code_sha256":"3f3c074604757720"}},{"arxiv_id":"2405.00218","paper":"/paper/constrained-decoding-for-secure-code","title":"Constrained Decoding for Secure Code Generation","date":"2024-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dynamite321/codeguardplus","path":"correctness_eval.py","file_url":"https://github.com/dynamite321/codeguardplus/blob/HEAD/correctness_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"aaf864304de22b06","mcp_get_code":{"code_sha256":"aaf864304de22b06"}},{"arxiv_id":"2402.09497","paper":"/paper/instruction-tuning-for-secure-code-generation","title":"Instruction Tuning for Secure Code Generation","date":"2024-02-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eth-sri/safecoder","path":"safecoder/metric.py","file_url":"https://github.com/eth-sri/safecoder/blob/HEAD/safecoder/metric.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3a7c7ec16769928a","mcp_get_code":{"code_sha256":"3a7c7ec16769928a"}},{"arxiv_id":"2306.10763","paper":"/paper/guiding-language-models-of-code-with-global","title":"Guiding Language Models of Code with Global Context using Monitors","date":"2023-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/monitors4codegen","path":"evaluation_scripts/eval_results.py","file_url":"https://github.com/microsoft/monitors4codegen/blob/HEAD/evaluation_scripts/eval_results.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"81302d4943b3cecc","mcp_get_code":{"code_sha256":"81302d4943b3cecc"}},{"arxiv_id":"2206.03865","paper":"/paper/fault-aware-neural-code-rankers","title":"Fault-Aware Neural Code Rankers","date":"2022-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/coderanker","path":"src/compute_metrics.py","file_url":"https://github.com/microsoft/coderanker/blob/HEAD/src/compute_metrics.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"62af43e66bbbc806","mcp_get_code":{"code_sha256":"62af43e66bbbc806"}},{"arxiv_id":"2107.03374","paper":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/PythonProgrammingPuzzles","path":"solvers/codex/run_codex_experiments.py","file_url":"https://github.com/microsoft/PythonProgrammingPuzzles/blob/HEAD/solvers/codex/run_codex_experiments.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2a310ceef0a63386","mcp_get_code":{"code_sha256":"2a310ceef0a63386"}}]}