{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/post-process","entry":"post_process","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":56,"n_papers_ran":31,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":36,"n_samples_ran":15,"n_samples_fingerprinted":9,"n_places":58,"n_places_pointer_only":15,"by_status":{"ran_honours":1,"ran_violates":1,"ran_draft_wrong":4,"ran_fixture":0,"ran":9,"unverified":21},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2604.10900","paper":"/paper/arxiv-2604-10900","title":"CASK: Core-Aware Selective KV Compression for Reasoning Traces","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"THUDM/LongBench","path":"LongBench/pred.py","file_url":"https://github.com/THUDM/LongBench/blob/HEAD/LongBench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2603.04759","paper":"/paper/arxiv-2603-04759","title":"Stacked from One: Multi-Scale Self-Injection for Context Window Extension","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Clement25/SharedLLM","path":"LongBenchTest/pred.py","file_url":"https://github.com/Clement25/SharedLLM/blob/HEAD/LongBenchTest/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2602.01267","paper":"/paper/arxiv-2602-01267","title":"Diving into Kronecker Adapters: Component Design Matters","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"rainstonee/CDKA","path":"eval_humaneval.py","file_url":"https://github.com/rainstonee/CDKA/blob/HEAD/eval_humaneval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"48720117c27620bc","mcp_get_code":{"code_sha256":"48720117c27620bc"}},{"arxiv_id":"2601.19232","paper":"/paper/arxiv-2601-19232","title":"Structure-based RNA Design by Step-wise Optimization of Latent Diffusion Model","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ml4bio/RiboDiffusion","path":"sampling.py","file_url":"https://github.com/ml4bio/RiboDiffusion/blob/HEAD/sampling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f06cabb9ba084d9c","mcp_get_code":{"code_sha256":"f06cabb9ba084d9c"}},{"arxiv_id":"2601.03042","paper":"/paper/arxiv-2601-03042","title":"BaseCal: Unsupervised Confidence Calibration via Base Model Signals","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Tan-Hexiang/BaseCal","path":"gen_metric/post_process.py","file_url":"https://github.com/Tan-Hexiang/BaseCal/blob/HEAD/gen_metric/post_process.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"569ffcd1c346d06c","mcp_get_code":{"code_sha256":"569ffcd1c346d06c"}},{"arxiv_id":"2509.19633","paper":"/paper/arxiv-2509-19633","title":"Mamba Modulation On the Length Generalization of Mamba","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"gnepul-ace/mamba_modulation","path":"Mamba/LongBench/pred.py","file_url":"https://github.com/gnepul-ace/mamba_modulation/blob/HEAD/Mamba/LongBench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2504.10479","paper":"/paper/internvl3-exploring-advanced-training-and","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","date":"2025-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opengvlab/internvl","path":"internvl_chat/eval/mmmu/evaluate_mmmu.py","file_url":"https://github.com/opengvlab/internvl/blob/HEAD/internvl_chat/eval/mmmu/evaluate_mmmu.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e9ddf26b2ed6ced5","mcp_get_code":{"code_sha256":"e9ddf26b2ed6ced5"}},{"arxiv_id":"2502.12052","paper":"/paper/a-dual-perspective-nlg-meta-evaluation","title":"A Dual-Perspective NLG Meta-Evaluation Framework with Automatic Benchmark and Better Interpretability","date":"2025-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"PKU-ONELab/NLG-DualEval","path":"module/data_process.py","file_url":"https://github.com/PKU-ONELab/NLG-DualEval/blob/HEAD/module/data_process.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5b373b7f1514f260","mcp_get_code":{"code_sha256":"5b373b7f1514f260"}},{"arxiv_id":"2501.18492","paper":"/paper/guardreasoner-towards-reasoning-based-llm","title":"GuardReasoner: Towards Reasoning-based LLM Safeguards","date":"2025-01-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yueliu1999/guardreasoner","path":"deploy.py","file_url":"https://github.com/yueliu1999/guardreasoner/blob/HEAD/deploy.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"758bc0c482412a97","mcp_get_code":{"code_sha256":"758bc0c482412a97"}},{"arxiv_id":"2411.17525","paper":"/paper/pushing-the-limits-of-large-language-model","title":"Pushing the Limits of Large Language Model Quantization via the Linearity Theorem","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"goodevening13/aquakv","path":"evaluate_longbench.py","file_url":"https://github.com/goodevening13/aquakv/blob/HEAD/evaluate_longbench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2410.14204","paper":"/paper/meditod-an-english-dialogue-dataset-for","title":"MediTOD: An English Dialogue Dataset for Medical History Taking with Comprehensive Annotations","date":"2024-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dair-iitd/MediTOD","path":"src/llama/post.py","file_url":"https://github.com/dair-iitd/MediTOD/blob/HEAD/src/llama/post.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"907ca7f3d23f9599","mcp_get_code":{"code_sha256":"907ca7f3d23f9599"}},{"arxiv_id":"2410.13846","paper":"/paper/simlayerkv-a-simple-framework-for-layer-level","title":"SimLayerKV: A Simple Framework for Layer-Level KV Cache Reduction","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/simlayerkv","path":"LongBench/pred.py","file_url":"https://github.com/sail-sg/simlayerkv/blob/HEAD/LongBench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2410.12381","paper":"/paper/humaneval-v-evaluating-visual-understanding","title":"HumanEval-V: Evaluating Visual Understanding and Reasoning Abilities of Large Multimodal Models Through Coding Tasks","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HumanEval-V/HumanEval-V-Benchmark","path":"evaluate.py","file_url":"https://github.com/HumanEval-V/HumanEval-V-Benchmark/blob/HEAD/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a917ba970738a94f","mcp_get_code":{"code_sha256":"a917ba970738a94f"}},{"arxiv_id":"2410.08811","paper":"/paper/poisonbench-assessing-large-language-model","title":"PoisonBench: Assessing Large Language Model Vulnerability to Data Poisoning","date":"2024-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TingchenFu/PoisonBench","path":"code/evaluation.py","file_url":"https://github.com/TingchenFu/PoisonBench/blob/HEAD/code/evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"487ba04abfcb1059","mcp_get_code":{"code_sha256":"487ba04abfcb1059"}},{"arxiv_id":"2410.05076","paper":"/paper/tidaldecode-fast-and-accurate-llm-decoding","title":"TidalDecode: Fast and Accurate LLM Decoding with Position Persistent Sparse Attention","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DerrickYLJ/TidalDecode","path":"experiments/LongBench/pred.py","file_url":"https://github.com/DerrickYLJ/TidalDecode/blob/HEAD/experiments/LongBench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2409.17422","paper":"/paper/discovering-the-gems-in-early-layers","title":"Discovering the Gems in Early Layers: Accelerating Long-Context LLMs with 1000x Input Token Reduction","date":"2024-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforceairesearch/gemfilter","path":"eval/LongBench/pred.py","file_url":"https://github.com/salesforceairesearch/gemfilter/blob/HEAD/eval/LongBench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2409.00138","paper":"/paper/privacylens-evaluating-privacy-norm-awareness","title":"PrivacyLens: Evaluating Privacy Norm Awareness of Language Models in Action","date":"2024-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SALT-NLP/PrivacyLens","path":"evaluation/get_final_action.py","file_url":"https://github.com/SALT-NLP/PrivacyLens/blob/HEAD/evaluation/get_final_action.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bc753da2fbd7dbb7","mcp_get_code":{"code_sha256":"bc753da2fbd7dbb7"}},{"arxiv_id":"2408.07092","paper":"/paper/post-training-sparse-attention-with-double","title":"Post-Training Sparse Attention with Double Sparsity","date":"2024-08-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"andy-yang-1/doublesparse","path":"LongBench/pred.py","file_url":"https://github.com/andy-yang-1/doublesparse/blob/HEAD/LongBench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2408.04556","paper":"/paper/bias-aware-low-rank-adaptation-mitigating","title":"BA-LoRA: Bias-Alleviating Low-Rank Adaptation to Mitigate Catastrophic Inheritance in Large Language Models","date":"2024-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cyp-jlu-ai/ba-lora","path":"inference/humaneval.py","file_url":"https://github.com/cyp-jlu-ai/ba-lora/blob/HEAD/inference/humaneval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"48720117c27620bc","mcp_get_code":{"code_sha256":"48720117c27620bc"}},{"arxiv_id":"2406.10774","paper":"/paper/quest-query-aware-sparsity-for-efficient-long","title":"Quest: Query-Aware Sparsity for Efficient Long-Context LLM Inference","date":"2024-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mit-han-lab/Quest","path":"evaluation/LongBench/pred.py","file_url":"https://github.com/mit-han-lab/Quest/blob/HEAD/evaluation/LongBench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2405.04434","paper":"/paper/deepseek-v2-a-strong-economical-and-efficient","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","date":"2024-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shadowpa0327/Palu","path":"run_long_bench.py","file_url":"https://github.com/shadowpa0327/Palu/blob/HEAD/run_long_bench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2404.14469","paper":"/paper/snapkv-llm-knows-what-you-are-looking-for","title":"SnapKV: LLM Knows What You are Looking for Before Generation","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fasterdecoding/snapkv","path":"experiments/LongBench/pred_snap.py","file_url":"https://github.com/fasterdecoding/snapkv/blob/HEAD/experiments/LongBench/pred_snap.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2404.11199","paper":"/paper/ribodiffusion-tertiary-structure-based-rna","title":"RiboDiffusion: Tertiary Structure-based RNA Inverse Folding with Generative Diffusion Models","date":"2024-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ml4bio/ribodiffusion","path":"sampling.py","file_url":"https://github.com/ml4bio/ribodiffusion/blob/HEAD/sampling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f06cabb9ba084d9c","mcp_get_code":{"code_sha256":"f06cabb9ba084d9c"}},{"arxiv_id":"2404.02893","paper":"/paper/chatglm-math-improving-math-problem-solving","title":"ChatGLM-Math: Improving Math Problem-Solving in Large Language Models with a Self-Critique Pipeline","date":"2024-04-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/chatglm-math","path":"judge.py","file_url":"https://github.com/thudm/chatglm-math/blob/HEAD/judge.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7d8ae032d1e62767","mcp_get_code":{"code_sha256":"7d8ae032d1e62767"}},{"arxiv_id":"2402.08679","paper":"/paper/cold-attack-jailbreaking-llms-with","title":"COLD-Attack: Jailbreaking LLMs with Stealthiness and Controllability","date":"2024-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yu-Fangxu/COLD-Attack","path":"evaluate.py","file_url":"https://github.com/Yu-Fangxu/COLD-Attack/blob/HEAD/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"84c0bec2c35fbb5b","mcp_get_code":{"code_sha256":"84c0bec2c35fbb5b"}},{"arxiv_id":"2402.02750","paper":"/paper/kivi-a-tuning-free-asymmetric-2bit","title":"KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jy-yuan/kivi","path":"pred_long_bench.py","file_url":"https://github.com/jy-yuan/kivi/blob/HEAD/pred_long_bench.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2401.02138","paper":"/paper/explore-human-parsing-modality-for-action-1","title":"Explore Human Parsing Modality for Action Recognition","date":"2024-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liujf69/EPP-Net-Action","path":"Parsing/View.py","file_url":"https://github.com/liujf69/EPP-Net-Action/blob/HEAD/Parsing/View.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6f1af45f036055b6","mcp_get_code":{"code_sha256":"6f1af45f036055b6"}},{"arxiv_id":"2312.00700","paper":"/paper/gift-generative-interpretable-fine-tuning","title":"Generative Parameter-Efficient Fine-Tuning","date":"2023-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"savadikarc/gift","path":"language_modeling/math_code_instruct/eval_humaneval.py","file_url":"https://github.com/savadikarc/gift/blob/HEAD/language_modeling/math_code_instruct/eval_humaneval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"48720117c27620bc","mcp_get_code":{"code_sha256":"48720117c27620bc"}},{"arxiv_id":"2311.18743","paper":"/paper/alignbench-benchmarking-chinese-alignment-of","title":"AlignBench: Benchmarking Chinese Alignment of Large Language Models","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/alignbench","path":"judge.py","file_url":"https://github.com/thudm/alignbench/blob/HEAD/judge.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7d8ae032d1e62767","mcp_get_code":{"code_sha256":"7d8ae032d1e62767"}},{"arxiv_id":"2311.09198","paper":"/paper/never-lost-in-the-middle-improving-large","title":"Never Lost in the Middle: Mastering Long-Context Question Answering with Position-Agnostic Decompositional Training","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hejunqing/never-lost-in-the-middle","path":"LongBench/src/pred_chatglm.py","file_url":"https://github.com/hejunqing/never-lost-in-the-middle/blob/HEAD/LongBench/src/pred_chatglm.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2311.08803","paper":"/paper/strategyllm-large-language-models-as-strategy","title":"StrategyLLM: Large Language Models as Strategy Generators, Executors, Optimizers, and Evaluators for Problem Solving","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gao-xiao-bai/StrategyLLM","path":"source/dataset/utils.py","file_url":"https://github.com/gao-xiao-bai/StrategyLLM/blob/HEAD/source/dataset/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8a20371953a7f5ec","mcp_get_code":{"code_sha256":"8a20371953a7f5ec"}},{"arxiv_id":"2311.08718","paper":"/paper/decomposing-uncertainty-for-large-language","title":"Decomposing Uncertainty for Large Language Models through Input Clarification Ensembling","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ucsb-nlp-chang/llm_uncertainty","path":"forward.py","file_url":"https://github.com/ucsb-nlp-chang/llm_uncertainty/blob/HEAD/forward.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a2f115a6ee40b1e0","mcp_get_code":{"code_sha256":"a2f115a6ee40b1e0"}},{"arxiv_id":"2310.10158","paper":"/paper/character-llm-a-trainable-agent-for-role","title":"Character-LLM: A Trainable Agent for Role-Playing","date":"2023-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"choosewhatulike/trainable-agents","path":"run_api_score_single.py","file_url":"https://github.com/choosewhatulike/trainable-agents/blob/HEAD/run_api_score_single.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7d2f2e6ced176d21","mcp_get_code":{"code_sha256":"7d2f2e6ced176d21"}},{"arxiv_id":"2310.08530","paper":"/paper/unipose-detecting-any-keypoints","title":"X-Pose: Detecting Any Keypoints","date":"2023-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IDEA-Research/UniPose","path":"models/UniPose/mask_generate.py","file_url":"https://github.com/IDEA-Research/UniPose/blob/HEAD/models/UniPose/mask_generate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"93ac42433372f73b","mcp_get_code":{"code_sha256":"93ac42433372f73b"}},{"arxiv_id":"2309.02233","paper":"/paper/augmenting-black-box-llms-with-medical","title":"Augmenting Black-box LLMs with Medical Textbooks for Biomedical Question Answering (Published in Findings of EMNLP 2024)","date":"2023-09-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TIGER-AI-Lab/LLM-AMT","path":"src/data_process/split_segments.py","file_url":"https://github.com/TIGER-AI-Lab/LLM-AMT/blob/HEAD/src/data_process/split_segments.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f2f92a62507bbd22","mcp_get_code":{"code_sha256":"f2f92a62507bbd22"}},{"arxiv_id":"2303.08774","paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AUCOHL/RTL-Repo","path":"src/utils.py","file_url":"https://github.com/AUCOHL/RTL-Repo/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"68968aa9dba46972","mcp_get_code":{"code_sha256":"68968aa9dba46972"}},{"arxiv_id":"2201.05609","paper":"/paper/multilingual-open-text-1-0-public-domain-news","title":"Multilingual Open Text Release 1: Public Domain News in 44 Languages","date":"2022-01-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VietHoang1512/khmer-nltk","path":"khmernltk/utils/data.py","file_url":"https://github.com/VietHoang1512/khmer-nltk/blob/HEAD/khmernltk/utils/data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"365063f7b2778eda","mcp_get_code":{"code_sha256":"365063f7b2778eda"}},{"arxiv_id":"2106.08746","paper":"/paper/real-time-attacks-against-deep-reinforcement","title":"Real-time Adversarial Perturbations against Deep Reinforcement Learning Policies: Attacks and Defenses","date":"2021-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ssg-research/ad3-action-distribution-divergence-detector","path":"src/agents/action_conditional_video_prediction.py","file_url":"https://github.com/ssg-research/ad3-action-distribution-divergence-detector/blob/HEAD/src/agents/action_conditional_video_prediction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3ba7c8da77c3ab15","mcp_get_code":{"code_sha256":"3ba7c8da77c3ab15"}},{"arxiv_id":"2106.08408","paper":"/paper/seeing-through-clouds-in-satellite-images","title":"Seeing Through Clouds in Satellite Images","date":"2021-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/farmvibes-ai","path":"ops/compute_cloud_prob/compute_cloud_prob.py","file_url":"https://github.com/microsoft/farmvibes-ai/blob/HEAD/ops/compute_cloud_prob/compute_cloud_prob.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3bd59bc7a1a36fa6","mcp_get_code":{"code_sha256":"3bd59bc7a1a36fa6"}},{"arxiv_id":"2105.14167","paper":"/paper/neurallog-natural-language-inference-with","title":"NeuralLog: Natural Language Inference with Joint Neural and Logical Reasoning","date":"2021-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eric11eca/NeuralLog","path":"src/udify_parser.py","file_url":"https://github.com/eric11eca/NeuralLog/blob/HEAD/src/udify_parser.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0d1a2a05ca2c64f9","mcp_get_code":{"code_sha256":"0d1a2a05ca2c64f9"}},{"arxiv_id":"2011.01536","paper":"/paper/transquest-translation-quality-estimation","title":"TransQuest: Translation Quality Estimation with Cross-lingual Transformers","date":"2020-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tharindudr/transQuest","path":"transquest/algo/word_level/microtransquest/format.py","file_url":"https://github.com/tharindudr/transQuest/blob/HEAD/transquest/algo/word_level/microtransquest/format.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"eff4679b609af119","mcp_get_code":{"code_sha256":"eff4679b609af119"}},{"arxiv_id":"2009.13891","paper":"/paper/towards-effective-context-for-meta","title":"Towards Effective Context for Meta-Reinforcement Learning: an Approach based on Contrastive Learning","date":"2020-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TJU-DRL-LAB/self-supervised-rl","path":"RL_with_Environment_Representation/ccm/plot_csv.py","file_url":"https://github.com/TJU-DRL-LAB/self-supervised-rl/blob/HEAD/RL_with_Environment_Representation/ccm/plot_csv.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a26c4433e3e83ea7","mcp_get_code":{"code_sha256":"a26c4433e3e83ea7"}},{"arxiv_id":"2007.03875","paper":"/paper/kqa-pro-a-large-diagnostic-dataset-for","title":"KQA Pro: A Dataset with Explicit Compositional Programs for Complex Question Answering over Knowledge Base","date":"2020-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shijx12/kqapro_baselines","path":"Bart_Program/predict.py","file_url":"https://github.com/shijx12/kqapro_baselines/blob/HEAD/Bart_Program/predict.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"facadc8602de5157","mcp_get_code":{"code_sha256":"facadc8602de5157"}},{"arxiv_id":"2005.13837","paper":"/paper/generating-diverse-and-consistent-qa-pairs","title":"Generating Diverse and Consistent QA pairs from Contexts with Information-Maximizing Hierarchical Conditional VAEs","date":"2020-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seanie12/Info-HCVAE","path":"vae/translate.py","file_url":"https://github.com/seanie12/Info-HCVAE/blob/HEAD/vae/translate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"845c524066f993b4","mcp_get_code":{"code_sha256":"845c524066f993b4"}},{"arxiv_id":"2004.08955","paper":"/paper/resnest-split-attention-networks","title":"ResNeSt: Split-Attention Networks","date":"2020-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"osmr/imgclsmob","path":"prep_model.py","file_url":"https://github.com/osmr/imgclsmob/blob/HEAD/prep_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8f535a7e6722bc55","mcp_get_code":{"code_sha256":"8f535a7e6722bc55"}},{"arxiv_id":"2002.00748","paper":"/paper/asking-questions-the-human-way-scalable","title":"Asking Questions the Human Way: Scalable Question-Answer Generation from Text Corpus","date":"2020-01-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bangliu/ACS-QG","path":"QG_postprocess_seq2seq.py","file_url":"https://github.com/bangliu/ACS-QG/blob/HEAD/QG_postprocess_seq2seq.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"04dfc0720567cfda","mcp_get_code":{"code_sha256":"04dfc0720567cfda"}},{"arxiv_id":"1912.02164","paper":"/paper/plug-and-play-language-models-a-simple","title":"Plug and Play Language Models: A Simple Approach to Controlled Text Generation","date":"2019-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fsoft-ailab/poem-generator","path":"ailamtho/utils/process.py","file_url":"https://github.com/fsoft-ailab/poem-generator/blob/HEAD/ailamtho/utils/process.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"49e475b822b30d82","mcp_get_code":{"code_sha256":"49e475b822b30d82"}},{"arxiv_id":"1908.11025","paper":"/paper/deep-floor-plan-recognition-using-a-multi","title":"Deep Floor Plan Recognition Using a Multi-Task Network with Room-Boundary-Guided Attention","date":"2019-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zcemycl/PyTorch-DeepFloorplan","path":"deploy.py","file_url":"https://github.com/zcemycl/PyTorch-DeepFloorplan/blob/HEAD/deploy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8933e685eed34b70","mcp_get_code":{"code_sha256":"8933e685eed34b70"}},{"arxiv_id":"1908.11025","paper":"/paper/deep-floor-plan-recognition-using-a-multi","title":"Deep Floor Plan Recognition Using a Multi-Task Network with Room-Boundary-Guided Attention","date":"2019-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zcemycl/TF2DeepFloorplan","path":"src/dfp/deploy.py","file_url":"https://github.com/zcemycl/TF2DeepFloorplan/blob/HEAD/src/dfp/deploy.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"9bea870bd4d520b3","mcp_get_code":{"code_sha256":"9bea870bd4d520b3"}},{"arxiv_id":"1907.09470","paper":"/paper/characterizing-attacks-on-deep-reinforcement","title":"Characterizing Attacks on Deep Reinforcement Learning","date":"2019-07-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ssg-research/flare","path":"src/agents/action_conditional_video_prediction.py","file_url":"https://github.com/ssg-research/flare/blob/HEAD/src/agents/action_conditional_video_prediction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3ba7c8da77c3ab15","mcp_get_code":{"code_sha256":"3ba7c8da77c3ab15"}},{"arxiv_id":"1901.04056","paper":"/paper/the-liver-tumor-segmentation-benchmark-lits","title":"The Liver Tumor Segmentation Benchmark (LiTS)","date":"2019-01-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HimBegginer/test_liver","path":"livermask/livermask.py","file_url":"https://github.com/HimBegginer/test_liver/blob/HEAD/livermask/livermask.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"b5d179c8f27c1792","mcp_get_code":{"code_sha256":"b5d179c8f27c1792"}},{"arxiv_id":"1901.02860","paper":"/paper/transformer-xl-attentive-language-models","title":"Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context","date":"2019-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Machine-Learning-Tokyo/Poetry-GAN","path":"lang_model.py","file_url":"https://github.com/Machine-Learning-Tokyo/Poetry-GAN/blob/HEAD/lang_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9d6735f22035b700","mcp_get_code":{"code_sha256":"9d6735f22035b700"}},{"arxiv_id":"aaai_34662","paper":null,"title":"arXiv:aaai_34662","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"maxindian/3D-RPE-Long-Contex-Modeling","path":"longbench-eval.py","file_url":"https://github.com/maxindian/3D-RPE-Long-Contex-Modeling/blob/HEAD/longbench-eval.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","paper":null,"title":"arXiv:Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenGVLab/PVC","path":"eval/mmbench/evaluate_mmbench.py","file_url":"https://github.com/OpenGVLab/PVC/blob/HEAD/eval/mmbench/evaluate_mmbench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e9ddf26b2ed6ced5","mcp_get_code":{"code_sha256":"e9ddf26b2ed6ced5"}},{"arxiv_id":"Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","paper":null,"title":"arXiv:Yang_PVC_Progressive_Visual_Token_Compression_for_Unified_Image_and_Video_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenGVLab/PVC","path":"eval/mmmu/evaluate_mmmu_cot.py","file_url":"https://github.com/OpenGVLab/PVC/blob/HEAD/eval/mmmu/evaluate_mmmu_cot.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3340b1a7ea1e9d7c","mcp_get_code":{"code_sha256":"3340b1a7ea1e9d7c"}},{"arxiv_id":"2025.emnlp-main.443","paper":null,"title":"arXiv:2025.emnlp-main.443","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"adlnlp/Gali","path":"longbench/pred.py","file_url":"https://github.com/adlnlp/Gali/blob/HEAD/longbench/pred.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4489113b536ca6eb","mcp_get_code":{"code_sha256":"4489113b536ca6eb"}},{"arxiv_id":"2023.findings-acl.892","paper":null,"title":"arXiv:2023.findings-acl.892","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"TharinduDR/TransQuest","path":"transquest/algo/word_level/microtransquest/format.py","file_url":"https://github.com/TharinduDR/TransQuest/blob/HEAD/transquest/algo/word_level/microtransquest/format.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"eff4679b609af119","mcp_get_code":{"code_sha256":"eff4679b609af119"}},{"arxiv_id":"2022.findings-emnlp.133","paper":null,"title":"arXiv:2022.findings-emnlp.133","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"shijx12/KQAPro_Baselines","path":"Bart_Program/predict.py","file_url":"https://github.com/shijx12/KQAPro_Baselines/blob/HEAD/Bart_Program/predict.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"facadc8602de5157","mcp_get_code":{"code_sha256":"facadc8602de5157"}}]}