{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-solutions","entry":"get_solutions","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":6,"n_papers_ran":1,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":2,"n_samples_ran":1,"n_samples_fingerprinted":0,"n_places":6,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2602.01685","paper":"/paper/arxiv-2602-01685","title":"Semantic-aware Wasserstein Policy Regularization for Large Language Model Alignment","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"aailab-kaist/WPR","path":"inference/testing_util.py","file_url":"https://github.com/aailab-kaist/WPR/blob/HEAD/inference/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0c83bb0f43b8dc57","mcp_get_code":{"code_sha256":"0c83bb0f43b8dc57"}},{"arxiv_id":"2403.04706","paper":"/paper/common-7b-language-models-already-possess","title":"Common 7B Language Models Already Possess Strong Math Capabilities","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xwin-lm/xwin-lm","path":"Xwin-Coder/APPS/testing_util.py","file_url":"https://github.com/xwin-lm/xwin-lm/blob/HEAD/Xwin-Coder/APPS/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0c83bb0f43b8dc57","mcp_get_code":{"code_sha256":"0c83bb0f43b8dc57"}},{"arxiv_id":"2310.19046","paper":"/paper/large-language-models-as-evolutionary","title":"Large Language Models as Evolutionary Optimizers","date":"2023-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cschen1205/lmea","path":"src/models/llm_tsp.py","file_url":"https://github.com/cschen1205/lmea/blob/HEAD/src/models/llm_tsp.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3c63b1c6daded93c","mcp_get_code":{"code_sha256":"3c63b1c6daded93c"}},{"arxiv_id":"2203.08597","paper":"/paper/less-is-more-summary-of-long-instructions-is","title":"Less is More: Summary of Long Instructions is Better for Program Synthesis","date":"2022-03-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kurbster/prompt-summarization","path":"src/lib/testing_util.py","file_url":"https://github.com/kurbster/prompt-summarization/blob/HEAD/src/lib/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0c83bb0f43b8dc57","mcp_get_code":{"code_sha256":"0c83bb0f43b8dc57"}},{"arxiv_id":"2107.03374","paper":"/paper/evaluating-large-language-models-trained-on","title":"Evaluating Large Language Models Trained on Code","date":"2021-07-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"codedotal/gpt-code-clippy","path":"evaluation/apps_eval_util.py","file_url":"https://github.com/codedotal/gpt-code-clippy/blob/HEAD/evaluation/apps_eval_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0c83bb0f43b8dc57","mcp_get_code":{"code_sha256":"0c83bb0f43b8dc57"}},{"arxiv_id":"2105.09938","paper":"/paper/measuring-coding-challenge-competence-with","title":"Measuring Coding Challenge Competence With APPS","date":"2021-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hendrycks/apps","path":"eval/testing_util.py","file_url":"https://github.com/hendrycks/apps/blob/HEAD/eval/testing_util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0c83bb0f43b8dc57","mcp_get_code":{"code_sha256":"0c83bb0f43b8dc57"}}]}