{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/run-benchmark","entry":"run_benchmark","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":11,"n_papers_ran":4,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":4,"n_samples_fingerprinted":0,"n_places":12,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":3,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2607.20630","paper":"/paper/arxiv-2607-20630","title":"Demonstrating GenDB: Instance-Optimized and Customized Query Processing Code Generation via LLM Agents","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"SolidLao/GenDB","path":"benchmarks/compare_gendb_versions.py","file_url":"https://github.com/SolidLao/GenDB/blob/HEAD/benchmarks/compare_gendb_versions.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a72bb96a49b6e674","mcp_get_code":{"code_sha256":"a72bb96a49b6e674"}},{"arxiv_id":"2607.20630","paper":"/paper/arxiv-2607-20630","title":"Demonstrating GenDB: Instance-Optimized and Customized Query Processing Code Generation via LLM Agents","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"SolidLao/GenDB","path":"benchmarks/compare_languages.py","file_url":"https://github.com/SolidLao/GenDB/blob/HEAD/benchmarks/compare_languages.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b6b1221fe542202c","mcp_get_code":{"code_sha256":"b6b1221fe542202c"}},{"arxiv_id":"2607.09371","paper":"/paper/arxiv-2607-09371","title":"Spectrally Deconfounded Gradient Boosting","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"fabsig/GPBoost","path":"external_libs/compute/perf/perf.py","file_url":"https://github.com/fabsig/GPBoost/blob/HEAD/external_libs/compute/perf/perf.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"603210f99dc97740","mcp_get_code":{"code_sha256":"603210f99dc97740"}},{"arxiv_id":"2506.05333","paper":"/paper/kinetics-rethinking-test-time-scaling-laws","title":"Kinetics: Rethinking Test-Time Scaling Laws","date":"2025-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"infini-ai-lab/kinetics","path":"benchmark/blocktopk.py","file_url":"https://github.com/infini-ai-lab/kinetics/blob/HEAD/benchmark/blocktopk.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dfaeb291c6e3d5a3","mcp_get_code":{"code_sha256":"dfaeb291c6e3d5a3"}},{"arxiv_id":"2506.01360","paper":null,"title":"arXiv:2506.01360","date":null,"month_inferred_from_arxiv_id":"2025-06","title_source":null,"repo":"chlehdwon/RDB2G-Bench","path":"rdb2g_bench/benchmark/bench_runner.py","file_url":"https://github.com/chlehdwon/RDB2G-Bench/blob/HEAD/rdb2g_bench/benchmark/bench_runner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e6b2981fb895ac29","mcp_get_code":{"code_sha256":"e6b2981fb895ac29"}},{"arxiv_id":"2502.11089","paper":"/paper/native-sparse-attention-hardware-aligned-and","title":"Native Sparse Attention: Hardware-Aligned and Natively Trainable Sparse Attention","date":"2025-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/seerattention","path":"eval/efficiency/benchmark_sparse_attn.py","file_url":"https://github.com/microsoft/seerattention/blob/HEAD/eval/efficiency/benchmark_sparse_attn.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"046184b390b18314","mcp_get_code":{"code_sha256":"046184b390b18314"}},{"arxiv_id":"2206.13424","paper":"/paper/benchopt-reproducible-efficient-and","title":"Benchopt: Reproducible, efficient and collaborative optimization benchmarks","date":"2022-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"benchopt/benchopt","path":"benchopt/runner.py","file_url":"https://github.com/benchopt/benchopt/blob/HEAD/benchopt/runner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"7143d36d09390525","mcp_get_code":{"code_sha256":"7143d36d09390525"}},{"arxiv_id":"2201.06239","paper":"/paper/mt-gbm-a-multi-task-gradient-boosting-machine-1","title":"MT-GBM: A Multi-Task Gradient Boosting Machine with Shared Decision Trees","date":"2022-01-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"antmachineintelligence/mtgbmcode","path":"compute/perf/perf.py","file_url":"https://github.com/antmachineintelligence/mtgbmcode/blob/HEAD/compute/perf/perf.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"603210f99dc97740","mcp_get_code":{"code_sha256":"603210f99dc97740"}},{"arxiv_id":"2105.08541","paper":"/paper/dacbench-a-benchmark-library-for-dynamic","title":"DACBench: A Benchmark Library for Dynamic Algorithm Configuration","date":"2021-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"automl/DACBench","path":"dacbench/runner.py","file_url":"https://github.com/automl/DACBench/blob/HEAD/dacbench/runner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"34e849e141cb91d1","mcp_get_code":{"code_sha256":"34e849e141cb91d1"}},{"arxiv_id":"1803.08823","paper":"/paper/a-high-bias-low-variance-introduction-to","title":"A high-bias, low-variance introduction to Machine Learning for physicists","date":"2018-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alexandreday/fast_density_clustering","path":"benchmarks/benchmark_rust_vs_python.py","file_url":"https://github.com/alexandreday/fast_density_clustering/blob/HEAD/benchmarks/benchmark_rust_vs_python.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"1b865279c9fa6b21","mcp_get_code":{"code_sha256":"1b865279c9fa6b21"}},{"arxiv_id":"1303.5778","paper":"/paper/speech-recognition-with-deep-recurrent-neural","title":"Speech Recognition with Deep Recurrent Neural Networks","date":"2013-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"1ytic/warp-rnnt","path":"pytorch_binding/benchmark.py","file_url":"https://github.com/1ytic/warp-rnnt/blob/HEAD/pytorch_binding/benchmark.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4c403a94d4f70757","mcp_get_code":{"code_sha256":"4c403a94d4f70757"}},{"arxiv_id":"2025.acl-long.191","paper":null,"title":"arXiv:2025.acl-long.191","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"UKPLab/acl2025-diverse-cot","path":"evaluation.py","file_url":"https://github.com/UKPLab/acl2025-diverse-cot/blob/HEAD/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5c072a2b08f9a815","mcp_get_code":{"code_sha256":"5c072a2b08f9a815"}}]}