{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/build-chat","entry":"build_chat","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":29,"n_papers_ran":19,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":23,"n_samples_ran":15,"n_samples_fingerprinted":1,"n_places":36,"n_places_pointer_only":11,"by_status":{"ran_honours":0,"ran_violates":1,"ran_draft_wrong":2,"ran_fixture":1,"ran":11,"unverified":8},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2604.10900","paper":"/paper/arxiv-2604-10900","title":"CASK: Core-Aware Selective KV Compression for Reasoning Traces","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"THUDM/LongBench","path":"LongBench/pred.py","file_url":"https://github.com/THUDM/LongBench/blob/HEAD/LongBench/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"324506640e971548","mcp_get_code":{"code_sha256":"324506640e971548"}},{"arxiv_id":"2604.10900","paper":"/paper/arxiv-2604-10900","title":"CASK: Core-Aware Selective KV Compression for Reasoning Traces","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"THUDM/LongBench","path":"LongBench/retrieval/pred.py","file_url":"https://github.com/THUDM/LongBench/blob/HEAD/LongBench/retrieval/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2e148fc130561394","mcp_get_code":{"code_sha256":"2e148fc130561394"}},{"arxiv_id":"2604.10900","paper":"/paper/arxiv-2604-10900","title":"CASK: Core-Aware Selective KV Compression for Reasoning Traces","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"THUDM/LongBench","path":"LongBench/summ/compress.py","file_url":"https://github.com/THUDM/LongBench/blob/HEAD/LongBench/summ/compress.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"04b0e165da9a0405","mcp_get_code":{"code_sha256":"04b0e165da9a0405"}},{"arxiv_id":"2603.04759","paper":"/paper/arxiv-2603-04759","title":"Stacked from One: Multi-Scale Self-Injection for Context Window Extension","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Clement25/SharedLLM","path":"LongBenchTest/summ/compress.py","file_url":"https://github.com/Clement25/SharedLLM/blob/HEAD/LongBenchTest/summ/compress.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"04b0e165da9a0405","mcp_get_code":{"code_sha256":"04b0e165da9a0405"}},{"arxiv_id":"2603.04759","paper":"/paper/arxiv-2603-04759","title":"Stacked from One: Multi-Scale Self-Injection for Context Window Extension","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Clement25/SharedLLM","path":"LongBenchTest/pred.py","file_url":"https://github.com/Clement25/SharedLLM/blob/HEAD/LongBenchTest/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9fc77302e79f7b27","mcp_get_code":{"code_sha256":"9fc77302e79f7b27"}},{"arxiv_id":"2509.19633","paper":"/paper/arxiv-2509-19633","title":"Mamba Modulation On the Length Generalization of Mamba","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"gnepul-ace/mamba_modulation","path":"Mamba/LongBench/pred.py","file_url":"https://github.com/gnepul-ace/mamba_modulation/blob/HEAD/Mamba/LongBench/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"324506640e971548","mcp_get_code":{"code_sha256":"324506640e971548"}},{"arxiv_id":"2506.08373","paper":"/paper/draft-based-approximate-inference-for-llms","title":"Draft-based Approximate Inference for LLMs","date":"2025-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"3ddd62fbad41e698","mcp_get_code":{"code_sha256":"3ddd62fbad41e698"}},{"arxiv_id":"2506.06409","paper":"/paper/heavywater-and-simplexwater-watermarking-low","title":"HeavyWater and SimplexWater: Watermarking Low-Entropy Text Distributions","date":"2025-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"f59d3f952a8d5950","mcp_get_code":{"code_sha256":"f59d3f952a8d5950"}},{"arxiv_id":"2505.23416","paper":"/paper/kvzip-query-agnostic-kv-cache-compression","title":"KVzip: Query-Agnostic KV Cache Compression with Context Reconstruction","date":"2025-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zefan-cai/kvcache-factory","path":"run_longbench.py","file_url":"https://github.com/zefan-cai/kvcache-factory/blob/HEAD/run_longbench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3ddd62fbad41e698","mcp_get_code":{"code_sha256":"3ddd62fbad41e698"}},{"arxiv_id":"2412.13670","paper":"/paper/antileak-bench-preventing-data-contamination","title":"AntiLeak-Bench: Preventing Data Contamination by Automatically Constructing Benchmarks with Updated Real-World Knowledge","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bobxwu/antileak-bench","path":"LLMs/LLM.py","file_url":"https://github.com/bobxwu/antileak-bench/blob/HEAD/LLMs/LLM.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"79f2a98626bab3df","mcp_get_code":{"code_sha256":"79f2a98626bab3df"}},{"arxiv_id":"2411.17525","paper":"/paper/pushing-the-limits-of-large-language-model","title":"Pushing the Limits of Large Language Model Quantization via the Linearity Theorem","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"goodevening13/aquakv","path":"evaluate_longbench.py","file_url":"https://github.com/goodevening13/aquakv/blob/HEAD/evaluate_longbench.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c3b06c2ebcf2af2d","mcp_get_code":{"code_sha256":"c3b06c2ebcf2af2d"}},{"arxiv_id":"2411.09688","paper":"/paper/squeezed-attention-accelerating-long-context","title":"Squeezed Attention: Accelerating Long Context Length LLM Inference","date":"2024-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SqueezeAILab/SqueezedAttention","path":"squeezedattention/utils.py","file_url":"https://github.com/SqueezeAILab/SqueezedAttention/blob/HEAD/squeezedattention/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"66d002e5d75a1496","mcp_get_code":{"code_sha256":"66d002e5d75a1496"}},{"arxiv_id":"2410.19258","paper":"/paper/not-all-heads-matter-a-head-level-kv-cache","title":"Not All Heads Matter: A Head-Level KV Cache Compression Method with Integrated Retrieval and Reasoning","date":"2024-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fyyfu/headkv","path":"run_longbench.py","file_url":"https://github.com/fyyfu/headkv/blob/HEAD/run_longbench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3ddd62fbad41e698","mcp_get_code":{"code_sha256":"3ddd62fbad41e698"}},{"arxiv_id":"2410.13846","paper":"/paper/simlayerkv-a-simple-framework-for-layer-level","title":"SimLayerKV: A Simple Framework for Layer-Level KV Cache Reduction","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/simlayerkv","path":"LongBench/pred.py","file_url":"https://github.com/sail-sg/simlayerkv/blob/HEAD/LongBench/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"324506640e971548","mcp_get_code":{"code_sha256":"324506640e971548"}},{"arxiv_id":"2410.05076","paper":"/paper/tidaldecode-fast-and-accurate-llm-decoding","title":"TidalDecode: Fast and Accurate LLM Decoding with Position Persistent Sparse Attention","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DerrickYLJ/TidalDecode","path":"experiments/LongBench/pred.py","file_url":"https://github.com/DerrickYLJ/TidalDecode/blob/HEAD/experiments/LongBench/pred.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1707cb62894075a9","mcp_get_code":{"code_sha256":"1707cb62894075a9"}},{"arxiv_id":"2409.17422","paper":"/paper/discovering-the-gems-in-early-layers","title":"Discovering the Gems in Early Layers: Accelerating Long-Context LLMs with 1000x Input Token Reduction","date":"2024-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforceairesearch/gemfilter","path":"eval/LongBench/pred.py","file_url":"https://github.com/salesforceairesearch/gemfilter/blob/HEAD/eval/LongBench/pred.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8339145374a403f3","mcp_get_code":{"code_sha256":"8339145374a403f3"}},{"arxiv_id":"2408.07092","paper":"/paper/post-training-sparse-attention-with-double","title":"Post-Training Sparse Attention with Double Sparsity","date":"2024-08-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"andy-yang-1/doublesparse","path":"LongBench/pred.py","file_url":"https://github.com/andy-yang-1/doublesparse/blob/HEAD/LongBench/pred.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c9d682400f2b2384","mcp_get_code":{"code_sha256":"c9d682400f2b2384"}},{"arxiv_id":"2407.13803","paper":"/paper/less-is-more-sparse-watermarking-in-llms-with","title":"Less is More: Sparse Watermarking in LLMs with Enhanced Text Quality","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mail-research/sparse-llm-watermarking","path":"pred.py","file_url":"https://github.com/mail-research/sparse-llm-watermarking/blob/HEAD/pred.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ee4afb2d89ea94b5","mcp_get_code":{"code_sha256":"ee4afb2d89ea94b5"}},{"arxiv_id":"2407.01527","paper":"/paper/kv-cache-compression-but-what-must-we-give-in","title":"KV Cache Compression, But What Must We Give in Return? A Comprehensive Benchmark of Long Context Capable Approaches","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"henryzhongsc/longctx_bench","path":"pipeline/model_utils.py","file_url":"https://github.com/henryzhongsc/longctx_bench/blob/HEAD/pipeline/model_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3b19a66bec30b1b9","mcp_get_code":{"code_sha256":"3b19a66bec30b1b9"}},{"arxiv_id":"2406.10774","paper":"/paper/quest-query-aware-sparsity-for-efficient-long","title":"Quest: Query-Aware Sparsity for Efficient Long-Context LLM Inference","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mit-han-lab/Quest","path":"evaluation/LongBench/pred.py","file_url":"https://github.com/mit-han-lab/Quest/blob/HEAD/evaluation/LongBench/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"324506640e971548","mcp_get_code":{"code_sha256":"324506640e971548"}},{"arxiv_id":"2406.07528","paper":"/paper/quickllama-query-aware-inference-acceleration","title":"QuickLLaMA: Query-aware Inference Acceleration for Large Language Models","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/q-llm","path":"benchmark/pred.py","file_url":"https://github.com/dvlab-research/q-llm/blob/HEAD/benchmark/pred.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"899f8b2f018b4018","mcp_get_code":{"code_sha256":"899f8b2f018b4018"}},{"arxiv_id":"2405.04434","paper":"/paper/deepseek-v2-a-strong-economical-and-efficient","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","date":"2024-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shadowpa0327/Palu","path":"run_long_bench.py","file_url":"https://github.com/shadowpa0327/Palu/blob/HEAD/run_long_bench.py","status":"ran_fixture","verification_level":1,"contract_check":"DEP_MISSING","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9d1a542959f8346c","mcp_get_code":{"code_sha256":"9d1a542959f8346c"}},{"arxiv_id":"2404.14469","paper":"/paper/snapkv-llm-knows-what-you-are-looking-for","title":"SnapKV: LLM Knows What You are Looking for Before Generation","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fasterdecoding/snapkv","path":"experiments/LongBench/pred_snap.py","file_url":"https://github.com/fasterdecoding/snapkv/blob/HEAD/experiments/LongBench/pred_snap.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ead90df220ef482e","mcp_get_code":{"code_sha256":"ead90df220ef482e"}},{"arxiv_id":"2404.04793","paper":"/paper/squeezeattention-2d-management-of-kv-cache-in","title":"SqueezeAttention: 2D Management of KV-Cache in LLM Inference via Layer-wise Optimal Budget","date":"2024-04-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hetailang/squeezeattention","path":"pred.py","file_url":"https://github.com/hetailang/squeezeattention/blob/HEAD/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2e148fc130561394","mcp_get_code":{"code_sha256":"2e148fc130561394"}},{"arxiv_id":"2402.04617","paper":"/paper/infllm-unveiling-the-intrinsic-capacity-of","title":"InfLLM: Training-Free Long-Context Extrapolation for LLMs with an Efficient Context Memory","date":"2024-02-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/infllm","path":"benchmark/pred.py","file_url":"https://github.com/thunlp/infllm/blob/HEAD/benchmark/pred.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"899f8b2f018b4018","mcp_get_code":{"code_sha256":"899f8b2f018b4018"}},{"arxiv_id":"2402.02750","paper":"/paper/kivi-a-tuning-free-asymmetric-2bit","title":"KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jy-yuan/kivi","path":"pred_long_bench.py","file_url":"https://github.com/jy-yuan/kivi/blob/HEAD/pred_long_bench.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ef82bf5af4441329","mcp_get_code":{"code_sha256":"ef82bf5af4441329"}},{"arxiv_id":"2311.09198","paper":"/paper/never-lost-in-the-middle-improving-large","title":"Never Lost in the Middle: Mastering Long-Context Question Answering with Position-Agnostic Decompositional Training","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hejunqing/never-lost-in-the-middle","path":"Lost_Retrieval/src/pred_single.py","file_url":"https://github.com/hejunqing/never-lost-in-the-middle/blob/HEAD/Lost_Retrieval/src/pred_single.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"251720ea85d2ca71","mcp_get_code":{"code_sha256":"251720ea85d2ca71"}},{"arxiv_id":"2311.09198","paper":"/paper/never-lost-in-the-middle-improving-large","title":"Never Lost in the Middle: Mastering Long-Context Question Answering with Position-Agnostic Decompositional Training","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hejunqing/never-lost-in-the-middle","path":"LongBench/src/baichuan_pred_shuffle.py","file_url":"https://github.com/hejunqing/never-lost-in-the-middle/blob/HEAD/LongBench/src/baichuan_pred_shuffle.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7024e12377186e51","mcp_get_code":{"code_sha256":"7024e12377186e51"}},{"arxiv_id":"2311.09198","paper":"/paper/never-lost-in-the-middle-improving-large","title":"Never Lost in the Middle: Mastering Long-Context Question Answering with Position-Agnostic Decompositional Training","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hejunqing/never-lost-in-the-middle","path":"LongBench/src/pred_chatglm.py","file_url":"https://github.com/hejunqing/never-lost-in-the-middle/blob/HEAD/LongBench/src/pred_chatglm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7439724189daf2cb","mcp_get_code":{"code_sha256":"7439724189daf2cb"}},{"arxiv_id":"2311.09198","paper":"/paper/never-lost-in-the-middle-improving-large","title":"Never Lost in the Middle: Mastering Long-Context Question Answering with Position-Agnostic Decompositional Training","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hejunqing/never-lost-in-the-middle","path":"LongBench/src/pred_shuffle_reader.py","file_url":"https://github.com/hejunqing/never-lost-in-the-middle/blob/HEAD/LongBench/src/pred_shuffle_reader.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0af745ab8b003692","mcp_get_code":{"code_sha256":"0af745ab8b003692"}},{"arxiv_id":"2311.07138","paper":"/paper/waterbench-towards-holistic-evaluation-of","title":"WaterBench: Towards Holistic Evaluation of Watermarks for Large Language Models","date":"2023-11-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"THU-KEG/WaterBench","path":"pred.py","file_url":"https://github.com/THU-KEG/WaterBench/blob/HEAD/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ed4be202f208766c","mcp_get_code":{"code_sha256":"ed4be202f208766c"}},{"arxiv_id":"2311.07138","paper":"/paper/waterbench-towards-holistic-evaluation-of","title":"WaterBench: Towards Holistic Evaluation of Watermarks for Large Language Models","date":"2023-11-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dortsur/heavywater_simplexwater","path":"pred.py","file_url":"https://github.com/dortsur/heavywater_simplexwater/blob/HEAD/pred.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f59d3f952a8d5950","mcp_get_code":{"code_sha256":"f59d3f952a8d5950"}},{"arxiv_id":"2307.15593","paper":"/paper/robust-distortion-free-watermarks-for","title":"Robust Distortion-free Watermarks for Language Models","date":"2023-07-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"f59d3f952a8d5950","mcp_get_code":{"code_sha256":"f59d3f952a8d5950"}},{"arxiv_id":"aaai_34662","paper":null,"title":"arXiv:aaai_34662","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"maxindian/3D-RPE-Long-Contex-Modeling","path":"longbench-eval.py","file_url":"https://github.com/maxindian/3D-RPE-Long-Contex-Modeling/blob/HEAD/longbench-eval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f59d3f952a8d5950","mcp_get_code":{"code_sha256":"f59d3f952a8d5950"}},{"arxiv_id":"2025.findings-emnlp.429","paper":null,"title":"arXiv:2025.findings-emnlp.429","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"zhzihao/QPruningKV","path":"run_longbench.py","file_url":"https://github.com/zhzihao/QPruningKV/blob/HEAD/run_longbench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3ddd62fbad41e698","mcp_get_code":{"code_sha256":"3ddd62fbad41e698"}},{"arxiv_id":"2025.emnlp-main.443","paper":null,"title":"arXiv:2025.emnlp-main.443","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"adlnlp/Gali","path":"longbench/pred.py","file_url":"https://github.com/adlnlp/Gali/blob/HEAD/longbench/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"324506640e971548","mcp_get_code":{"code_sha256":"324506640e971548"}}]}