{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/sha256-file","entry":"sha256_file","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":21,"n_papers_ran":14,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":20,"n_samples_ran":11,"n_samples_fingerprinted":0,"n_places":23,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":10,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.13279","paper":"/paper/arxiv-2609-13279","title":"The MODA General Attribute Suite: A Four-Track Evaluation Benchmark for Fashion Attribute Extraction","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"hopit-ai/Moda_ner","path":"suite/catalog/score.py","file_url":"https://github.com/hopit-ai/Moda_ner/blob/HEAD/suite/catalog/score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cb1f4090fbd1ecb5","mcp_get_code":{"code_sha256":"cb1f4090fbd1ecb5"}},{"arxiv_id":"2609.13279","paper":"/paper/arxiv-2609-13279","title":"The MODA General Attribute Suite: A Four-Track Evaluation Benchmark for Fashion Attribute Extraction","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"hopit-ai/Moda_ner","path":"suite/_model/package.py","file_url":"https://github.com/hopit-ai/Moda_ner/blob/HEAD/suite/_model/package.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7c7dd1b82c44d4f","mcp_get_code":{"code_sha256":"c7c7dd1b82c44d4f"}},{"arxiv_id":"2609.04105","paper":"/paper/arxiv-2609-04105","title":"Hardware-Aware FP4 FlashAttention-4","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"MrHuff/fp4-fa4","path":"results/fp4_fa4_technical_report_v2_20260819/export_llama8b_b4_snapshot.py","file_url":"https://github.com/MrHuff/fp4-fa4/blob/HEAD/results/fp4_fa4_technical_report_v2_20260819/export_llama8b_b4_snapshot.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cb1f4090fbd1ecb5","mcp_get_code":{"code_sha256":"cb1f4090fbd1ecb5"}},{"arxiv_id":"2609.02548","paper":"/paper/arxiv-2609-02548","title":"Learn from Whoever Is Right: Answer-Verified Multi-Teacher Distillation for Multi-Domain LLMs","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"hexixiang/MT-SDPO","path":"evaluation/evaluate.py","file_url":"https://github.com/hexixiang/MT-SDPO/blob/HEAD/evaluation/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5ab684253f906180","mcp_get_code":{"code_sha256":"5ab684253f906180"}},{"arxiv_id":"2609.01814","paper":"/paper/arxiv-2609-01814","title":"When Does Information Sharing Improve Decentralized Discovery? Aggregation, Independent Rescue, and Equilibrium Selection","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"yoheinakajima/distributed-discovery","path":"src/distributed_discovery/agent_ops/core.py","file_url":"https://github.com/yoheinakajima/distributed-discovery/blob/HEAD/src/distributed_discovery/agent_ops/core.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e87e452c60fef107","mcp_get_code":{"code_sha256":"e87e452c60fef107"}},{"arxiv_id":"2608.28281","paper":"/paper/arxiv-2608-28281","title":"LoopArena: Benchmarking Models as Runtime Controllers for Loop Engineering","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"AMAP-ML/LoopArena","path":"src/looparena/benchmarks/type3.py","file_url":"https://github.com/AMAP-ML/LoopArena/blob/HEAD/src/looparena/benchmarks/type3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a8c21f6c4ada713e","mcp_get_code":{"code_sha256":"a8c21f6c4ada713e"}},{"arxiv_id":"2608.25655","paper":"/paper/arxiv-2608-25655","title":"Reconstructing the Right Episode: Evaluating Interleaved Conversational Memory Beyond Long Context","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"LordTARN1SHED/SCALE-QA","path":"supplementary/noise_assets/noise_io.py","file_url":"https://github.com/LordTARN1SHED/SCALE-QA/blob/HEAD/supplementary/noise_assets/noise_io.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"f103025411f50529","mcp_get_code":{"code_sha256":"f103025411f50529"}},{"arxiv_id":"2608.24067","paper":"/paper/arxiv-2608-24067","title":"A Feature-Major Codebook for Memory-Efficient Sparse-Binary Self-Organizing Maps: Scaling a MEDLINE Atlas to 1.05 Million Neurons on a Single Consumer GPU","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"mongrolwarrior/sparsesom-paper1","path":"pipeline/fetch_data.py","file_url":"https://github.com/mongrolwarrior/sparsesom-paper1/blob/HEAD/pipeline/fetch_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7b16c59a7518bdd5","mcp_get_code":{"code_sha256":"7b16c59a7518bdd5"}},{"arxiv_id":"2608.13006","paper":"/paper/arxiv-2608-13006","title":"EviReform: Evidence-Guided Query Reformulation for Multi-Hop Graph Retrieval","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"XrazyMee/EviReform","path":"src/evireform/embeddings.py","file_url":"https://github.com/XrazyMee/EviReform/blob/HEAD/src/evireform/embeddings.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cb1f4090fbd1ecb5","mcp_get_code":{"code_sha256":"cb1f4090fbd1ecb5"}},{"arxiv_id":"2608.10896","paper":"/paper/arxiv-2608-10896","title":"Self-Normalized Inference for Constant-Stepsize Temporal-Difference Learning under Markovian Sampling","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"MinZenggit/SN-TD","path":"code/Experiments/garnet_policy_evaluation/io_utils.py","file_url":"https://github.com/MinZenggit/SN-TD/blob/HEAD/code/Experiments/garnet_policy_evaluation/io_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3b4b4b882b5a55cd","mcp_get_code":{"code_sha256":"3b4b4b882b5a55cd"}},{"arxiv_id":"2608.10288","paper":"/paper/arxiv-2608-10288","title":"Power law graph attention: exact generalization of scaled dot-product attention, empirical collapse at inference","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"burcgokden/PLDR-LLM-Math-Foundations","path":"audit/audit_online_lib.py","file_url":"https://github.com/burcgokden/PLDR-LLM-Math-Foundations/blob/HEAD/audit/audit_online_lib.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd09bc0035d76c74","mcp_get_code":{"code_sha256":"cd09bc0035d76c74"}},{"arxiv_id":"2608.09988","paper":"/paper/arxiv-2608-09988","title":"OpenPM: Auditable Point-in-Time Evaluation for LLM Portfolio-Management Agents","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"aslcai/OpenPM-Bench","path":"collectors/freeze.py","file_url":"https://github.com/aslcai/OpenPM-Bench/blob/HEAD/collectors/freeze.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8862dc4ae1fa2b0f","mcp_get_code":{"code_sha256":"8862dc4ae1fa2b0f"}},{"arxiv_id":"2607.25886","paper":"/paper/arxiv-2607-25886","title":"RSIBench-Data: Benchmarking Data-Centric Research for Recursive Self-Improvement","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"evolvent-ai/RSIBench-Data","path":"runner/lib/audit_agent_tamper.py","file_url":"https://github.com/evolvent-ai/RSIBench-Data/blob/HEAD/runner/lib/audit_agent_tamper.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb1f4090fbd1ecb5","mcp_get_code":{"code_sha256":"cb1f4090fbd1ecb5"}},{"arxiv_id":"2606.03938","paper":"/paper/arxiv-2606-03938","title":"q0: Primitives for Hyper-Epoch Pretraining","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"qlabs-eng/slowrun","path":"prepare_data.py","file_url":"https://github.com/qlabs-eng/slowrun/blob/HEAD/prepare_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5569e16cd0139606","mcp_get_code":{"code_sha256":"5569e16cd0139606"}},{"arxiv_id":"2606.00579","paper":"/paper/arxiv-2606-00579","title":"Sandboxed Coding Agents are Competitive Omni-modal Task Solvers","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Dongping-Chen/OmniCoding","path":"src/omnicoding/benchmarks/subsets.py","file_url":"https://github.com/Dongping-Chen/OmniCoding/blob/HEAD/src/omnicoding/benchmarks/subsets.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"5789280e1f024de4","mcp_get_code":{"code_sha256":"5789280e1f024de4"}},{"arxiv_id":"2606.00329","paper":"/paper/arxiv-2606-00329","title":"Benchmarking Recursive-Collapse Warning Claims Under Matched False-Positive Control","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"davidmullett/loopzero-paper-public","path":"src/loopzero_paper/benchmarks/recommender/bridge_check.py","file_url":"https://github.com/davidmullett/loopzero-paper-public/blob/HEAD/src/loopzero_paper/benchmarks/recommender/bridge_check.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d8fa2b5dd561cd7b","mcp_get_code":{"code_sha256":"d8fa2b5dd561cd7b"}},{"arxiv_id":"2603.15020","paper":"/paper/arxiv-2603-15020","title":"MER-Bench: A Comprehensive Benchmark for Multimodal Meme Reappraisal","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"one-seven17/MER-Bench","path":"judge.py","file_url":"https://github.com/one-seven17/MER-Bench/blob/HEAD/judge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"653b01dee433dd03","mcp_get_code":{"code_sha256":"653b01dee433dd03"}},{"arxiv_id":"2603.05764","paper":"/paper/arxiv-2603-05764","title":"TML-bench: Benchmark for Data Science Agents on Tabular ML Tasks","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"MykolaPinchuk/TML-bench","path":"orchestrator/hash_utils.py","file_url":"https://github.com/MykolaPinchuk/TML-bench/blob/HEAD/orchestrator/hash_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d106646067c3cd11","mcp_get_code":{"code_sha256":"d106646067c3cd11"}},{"arxiv_id":"2603.05764","paper":"/paper/arxiv-2603-05764","title":"TML-bench: Benchmark for Data Science Agents on Tabular ML Tasks","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"MykolaPinchuk/TML-bench","path":"orchestrator/provenance.py","file_url":"https://github.com/MykolaPinchuk/TML-bench/blob/HEAD/orchestrator/provenance.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8d12fe7f3b8fd20d","mcp_get_code":{"code_sha256":"8d12fe7f3b8fd20d"}},{"arxiv_id":"2505.23281","paper":"/paper/matharena-evaluating-llms-on-uncontaminated","title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","date":"2025-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eth-sri/matharena","path":"src/matharena/arxivmath_source.py","file_url":"https://github.com/eth-sri/matharena/blob/HEAD/src/matharena/arxivmath_source.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"66fa4b8fdf3fcdfb","mcp_get_code":{"code_sha256":"66fa4b8fdf3fcdfb"}},{"arxiv_id":"2409.13449","paper":"/paper/minstrel-structural-prompt-generation-with","title":"Minstrel: Structural Prompt Generation with Multi-Agents Coordination for Non-AI Experts","date":"2024-09-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sci-m-wang/minstrel","path":"src/sideprofile/bundle.py","file_url":"https://github.com/sci-m-wang/minstrel/blob/HEAD/src/sideprofile/bundle.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1933ce222ae6a0b8","mcp_get_code":{"code_sha256":"1933ce222ae6a0b8"}},{"arxiv_id":"2309.05794","paper":"/paper/diffusion-based-adversarial-purification-for","title":"Robust Physics-based Deep MRI Reconstruction Via Diffusion Purification","date":"2023-09-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sjames40/adversarial-purification-for-mri","path":"rodio_purify.py","file_url":"https://github.com/sjames40/adversarial-purification-for-mri/blob/HEAD/rodio_purify.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3abeaeb495c8f003","mcp_get_code":{"code_sha256":"3abeaeb495c8f003"}},{"arxiv_id":"2206.04670","paper":"/paper/pointnext-revisiting-pointnet-with-improved","title":"PointNeXt: Revisiting PointNet++ with Improved Training and Scaling Strategies","date":"2022-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guochengqian/pointnext","path":"pointnext_official/checkpoints.py","file_url":"https://github.com/guochengqian/pointnext/blob/HEAD/pointnext_official/checkpoints.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7f2983739479f50d","mcp_get_code":{"code_sha256":"7f2983739479f50d"}}]}