{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-questions","entry":"load_questions","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":17,"n_papers_ran":9,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":13,"n_samples_ran":5,"n_samples_fingerprinted":0,"n_places":17,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":5,"unverified":8},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.10528","paper":"/paper/arxiv-2606-10528","title":"Representation-Aware Advantage Estimation: Your Reward Model Provides More Than A Scalar Output","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"lmarena/arena-hard-auto","path":"utils/completion.py","file_url":"https://github.com/lmarena/arena-hard-auto/blob/HEAD/utils/completion.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"47892c5631a199d4","mcp_get_code":{"code_sha256":"47892c5631a199d4"}},{"arxiv_id":"2605.15482","paper":"/paper/arxiv-2605-15482","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"LimexAILab/FINESSE-Bench","path":"utils/completion.py","file_url":"https://github.com/LimexAILab/FINESSE-Bench/blob/HEAD/utils/completion.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"e88bc8fb3f14d5ab","mcp_get_code":{"code_sha256":"e88bc8fb3f14d5ab"}},{"arxiv_id":"2604.11048","paper":"/paper/arxiv-2604-11048","title":"A Systematic Analysis of the Impact of Persona Steering on LLM Capabilities","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"cjia7/DPR","path":"src/npti/neuron/apply_neuron_steering.py","file_url":"https://github.com/cjia7/DPR/blob/HEAD/src/npti/neuron/apply_neuron_steering.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e2da9069e83fab5e","mcp_get_code":{"code_sha256":"e2da9069e83fab5e"}},{"arxiv_id":"2603.20821","paper":"/paper/arxiv-2603-20821","title":"Compass: Optimizing Compound AI Workflows for Dynamic Adaptation","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"polaris-slo-cloud/compass","path":"experiments/run_baseline.py","file_url":"https://github.com/polaris-slo-cloud/compass/blob/HEAD/experiments/run_baseline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"75cfc0af9f628d6d","mcp_get_code":{"code_sha256":"75cfc0af9f628d6d"}},{"arxiv_id":"2603.01990","paper":"/paper/arxiv-2603-01990","title":"According to Me: Long-Term Personalized Referential Memory QA","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"JingbiaoMei/ATM-Bench","path":"agent_systems/prepare_sandbox.py","file_url":"https://github.com/JingbiaoMei/ATM-Bench/blob/HEAD/agent_systems/prepare_sandbox.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e748b717c3a734f2","mcp_get_code":{"code_sha256":"e748b717c3a734f2"}},{"arxiv_id":"2602.12196","paper":"/paper/arxiv-2602-12196","title":"Visual Reasoning Benchmark: Evaluating Multimodal LLMs on Classroom-Authentic Visual Problems from Primary Education","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"AI-for-Education/vrb-benchmark","path":"src/vrb_benchmark/load.py","file_url":"https://github.com/AI-for-Education/vrb-benchmark/blob/HEAD/src/vrb_benchmark/load.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d037d25ed74d5ca1","mcp_get_code":{"code_sha256":"d037d25ed74d5ca1"}},{"arxiv_id":"2602.10718","paper":"/paper/arxiv-2602-10718","title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"meituan-longcat/SGLang-FluentLLM","path":"benchmark/mtbench/bench_sglang.py","file_url":"https://github.com/meituan-longcat/SGLang-FluentLLM/blob/HEAD/benchmark/mtbench/bench_sglang.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f5ada5a1db00967b","mcp_get_code":{"code_sha256":"f5ada5a1db00967b"}},{"arxiv_id":"2507.09580","paper":null,"title":"arXiv:2507.09580","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"wangyu-ovo/aicrypto-agent","path":"run_choice_question.py","file_url":"https://github.com/wangyu-ovo/aicrypto-agent/blob/HEAD/run_choice_question.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1a1dce904cebb172","mcp_get_code":{"code_sha256":"1a1dce904cebb172"}},{"arxiv_id":"2410.21352","paper":"/paper/llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AboveParadise/LLMCBench","path":"tqa_utilities.py","file_url":"https://github.com/AboveParadise/LLMCBench/blob/HEAD/tqa_utilities.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a58462e59e01f858","mcp_get_code":{"code_sha256":"a58462e59e01f858"}},{"arxiv_id":"2410.11623","paper":"/paper/videgothink-assessing-egocentric-video","title":"VidEgoThink: Assessing Egocentric Video Understanding Capabilities for Embodied AI","date":"2024-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"adacheng/egothink","path":"common.py","file_url":"https://github.com/adacheng/egothink/blob/HEAD/common.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"88969cdc675c418e","mcp_get_code":{"code_sha256":"88969cdc675c418e"}},{"arxiv_id":"2409.15254","paper":"/paper/archon-an-architecture-search-framework-for","title":"Archon: An Architecture Search Framework for Inference-Time Techniques","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scalingintelligence/archon","path":"src/archon/benchmarks/arena_hard_auto/arena_hard_auto_utils.py","file_url":"https://github.com/scalingintelligence/archon/blob/HEAD/src/archon/benchmarks/arena_hard_auto/arena_hard_auto_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"47892c5631a199d4","mcp_get_code":{"code_sha256":"47892c5631a199d4"}},{"arxiv_id":"2408.08435","paper":"/paper/automated-design-of-agentic-systems","title":"Automated Design of Agentic Systems","date":"2024-08-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shengranhu/adas","path":"_gpqa/utils.py","file_url":"https://github.com/shengranhu/adas/blob/HEAD/_gpqa/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7b1e70edd9eaffe7","mcp_get_code":{"code_sha256":"7b1e70edd9eaffe7"}},{"arxiv_id":"2407.10972","paper":"/paper/vgbench-evaluating-large-language-models-on","title":"VGBench: Evaluating Large Language Models on Vector Graphics Understanding and Generation","date":"2024-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vgbench/VGBench","path":"build_dataset.py","file_url":"https://github.com/vgbench/VGBench/blob/HEAD/build_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d54d6d247c2fb8dc","mcp_get_code":{"code_sha256":"d54d6d247c2fb8dc"}},{"arxiv_id":"2406.11939","paper":"/paper/from-crowdsourced-data-to-high-quality","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lm-sys/arena-hard-auto","path":"utils/completion.py","file_url":"https://github.com/lm-sys/arena-hard-auto/blob/HEAD/utils/completion.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"47892c5631a199d4","mcp_get_code":{"code_sha256":"47892c5631a199d4"}},{"arxiv_id":"2309.16240","paper":"/paper/beyond-reverse-kl-generalizing-direct","title":"Beyond Reverse KL: Generalizing Direct Preference Optimization with Diverse Divergence Constraints","date":"2023-09-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alecwangcq/f-divergence-dpo","path":"mt_bench/common.py","file_url":"https://github.com/alecwangcq/f-divergence-dpo/blob/HEAD/mt_bench/common.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"88969cdc675c418e","mcp_get_code":{"code_sha256":"88969cdc675c418e"}},{"arxiv_id":"2025.findings-acl.979","paper":null,"title":"arXiv:2025.findings-acl.979","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Mrshenshen/FRUIT","path":"common.py","file_url":"https://github.com/Mrshenshen/FRUIT/blob/HEAD/common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fab3c2e9c03bc0c5","mcp_get_code":{"code_sha256":"fab3c2e9c03bc0c5"}},{"arxiv_id":"2025.acl-long.851","paper":null,"title":"arXiv:2025.acl-long.851","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"uw-nsl/SafeDecoding","path":"mt_bench/common.py","file_url":"https://github.com/uw-nsl/SafeDecoding/blob/HEAD/mt_bench/common.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"88969cdc675c418e","mcp_get_code":{"code_sha256":"88969cdc675c418e"}}]}