{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/is-number","entry":"is_number","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":70,"n_papers_ran":53,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":21,"n_samples_ran":13,"n_samples_fingerprinted":1,"n_places":70,"n_places_pointer_only":28,"by_status":{"ran_honours":0,"ran_violates":5,"ran_draft_wrong":0,"ran_fixture":0,"ran":8,"unverified":8},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.17360","paper":"/paper/arxiv-2608-17360","title":"Fair ASR: Re-Evaluating Black-Box Jailbreaks under Shared Target-Call Budgets","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"xsddys/Fair-ASR","path":"FairASR/evaluators/score_history_strong_reject_budget_global.py","file_url":"https://github.com/xsddys/Fair-ASR/blob/HEAD/FairASR/evaluators/score_history_strong_reject_budget_global.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6961743def4c7c57","mcp_get_code":{"code_sha256":"6961743def4c7c57"}},{"arxiv_id":"2604.23264","paper":"/paper/arxiv-2604-23264","title":"MotionHiFlow: Text-to-Motion via Hierarchical Flow Matching","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ai-lh/MotionHiFlow","path":"src/utils/get_opt.py","file_url":"https://github.com/ai-lh/MotionHiFlow/blob/HEAD/src/utils/get_opt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e9840a10d733a764","mcp_get_code":{"code_sha256":"e9840a10d733a764"}},{"arxiv_id":"2604.18124","paper":"/paper/arxiv-2604-18124","title":"TLoRA: Task-aware Low Rank Adaptation of Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Rambo-Yi/TLora","path":"evaluate/math/eval_gsm8k.py","file_url":"https://github.com/Rambo-Yi/TLora/blob/HEAD/evaluate/math/eval_gsm8k.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2604.11056","paper":"/paper/arxiv-2604-11056","title":"Where Hindsight Credit Can Reside: A Signed-Capacity View of Token Updates in RLVR","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"QwenLM/Qwen2.5-Math","path":"evaluation/math_utils.py","file_url":"https://github.com/QwenLM/Qwen2.5-Math/blob/HEAD/evaluation/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1c1060f6e43d5dfa","mcp_get_code":{"code_sha256":"1c1060f6e43d5dfa"}},{"arxiv_id":"2603.20017","paper":"/paper/arxiv-2603-20017","title":"RouterKGQA: Specialized-General Model Routing for Constraint-Aware Knowledge Graph Question Answering","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Oldcircle/RouterKGQA","path":"evaluate.py","file_url":"https://github.com/Oldcircle/RouterKGQA/blob/HEAD/evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b2feca00ae40a160","mcp_get_code":{"code_sha256":"b2feca00ae40a160"}},{"arxiv_id":"2603.02913","paper":"/paper/arxiv-2603-02913","title":"Eliciting Numerical Predictive Distributions of LLMs Without Autoregression","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"kasia-kobalczyk/guess_llm","path":"src/guess_llm/llm_utils/llm_no_scaling.py","file_url":"https://github.com/kasia-kobalczyk/guess_llm/blob/HEAD/src/guess_llm/llm_utils/llm_no_scaling.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2510.06307","paper":"/paper/arxiv-2510-06307","title":"Belief-Calibrated Multi-Agent Consensus Seeking for Complex NLP Tasks","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"dengwentao99/BCCS","path":"MATH/math_utils.py","file_url":"https://github.com/dengwentao99/BCCS/blob/HEAD/MATH/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1c1060f6e43d5dfa","mcp_get_code":{"code_sha256":"1c1060f6e43d5dfa"}},{"arxiv_id":"2505.19716","paper":"/paper/concise-reasoning-big-gains-pruning-long","title":"Concise Reasoning, Big Gains: Pruning Long Reasoning Trace with Difficulty-Aware Prompting","date":"2025-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"evanwu1125/litecot","path":"eval/Qwen2.5-Math/evaluation/math_utils.py","file_url":"https://github.com/evanwu1125/litecot/blob/HEAD/eval/Qwen2.5-Math/evaluation/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1c1060f6e43d5dfa","mcp_get_code":{"code_sha256":"1c1060f6e43d5dfa"}},{"arxiv_id":"2505.12371","paper":"/paper/medagentboard-benchmarking-multi-agent","title":"MedAgentBoard: Benchmarking Multi-Agent Collaboration with Conventional Methods for Diverse Medical Tasks","date":"2025-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HAIRLAB/Pre_Surv_COVID_19","path":"utils.py","file_url":"https://github.com/HAIRLAB/Pre_Surv_COVID_19/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f97aabe235dec048","mcp_get_code":{"code_sha256":"f97aabe235dec048"}},{"arxiv_id":"2504.20571","paper":"/paper/reinforcement-learning-for-reasoning-in-large","title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","date":"2025-04-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ypwang61/one-shot-rlvr","path":"Qwen2.5-Eval/evaluation/math_utils.py","file_url":"https://github.com/ypwang61/one-shot-rlvr/blob/HEAD/Qwen2.5-Eval/evaluation/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1c1060f6e43d5dfa","mcp_get_code":{"code_sha256":"1c1060f6e43d5dfa"}},{"arxiv_id":"2504.05185","paper":"/paper/concise-reasoning-via-reinforcement-learning","title":"Concise Reasoning via Reinforcement Learning","date":"2025-04-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ai-wand/concise-reasoning","path":"eval/math_utils.py","file_url":"https://github.com/ai-wand/concise-reasoning/blob/HEAD/eval/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1c1060f6e43d5dfa","mcp_get_code":{"code_sha256":"1c1060f6e43d5dfa"}},{"arxiv_id":"2502.11133","paper":"/paper/masrouter-learning-to-route-llms-for-multi","title":"MasRouter: Learning to Route LLMs for Multi-Agent Systems","date":"2025-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yanweiyue/masrouter","path":"Datasets/gsm8k/gsm8k_dataset.py","file_url":"https://github.com/yanweiyue/masrouter/blob/HEAD/Datasets/gsm8k/gsm8k_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4f015b339ccf7dee","mcp_get_code":{"code_sha256":"4f015b339ccf7dee"}},{"arxiv_id":"2501.12599","paper":"/paper/kimi-k1-5-scaling-reinforcement-learning-with","title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","date":"2025-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/math-v","path":"models/utils.py","file_url":"https://github.com/mathllm/math-v/blob/HEAD/models/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2501.03012","paper":"/paper/analyzing-fine-tuning-representation-shift","title":"Analyzing Fine-tuning Representation Shift for Multimodal LLMs Steering alignment","date":"2025-01-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mshukor/xl-vlms","path":"src/analysis/cluster_analysis.py","file_url":"https://github.com/mshukor/xl-vlms/blob/HEAD/src/analysis/cluster_analysis.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3b5feddb8481ce67","mcp_get_code":{"code_sha256":"3b5feddb8481ce67"}},{"arxiv_id":"2412.18537","paper":"/paper/harnessing-large-language-models-for-1","title":"Harnessing Large Language Models for Knowledge Graph Question Answering via Adaptive Multi-Aspect Retrieval-Augmentation","date":"2024-12-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Applied-Machine-Learning-Lab/AMAR","path":"eval_final.py","file_url":"https://github.com/Applied-Machine-Learning-Lab/AMAR/blob/HEAD/eval_final.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b2feca00ae40a160","mcp_get_code":{"code_sha256":"b2feca00ae40a160"}},{"arxiv_id":"2412.11006","paper":"/paper/entropy-regularized-process-reward-model","title":"Entropy-Regularized Process Reward Model","date":"2024-12-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2410.09344","paper":"/paper/dare-the-extreme-revisiting-delta-parameter","title":"DARE the Extreme: Revisiting Delta-Parameter Pruning For Fine-Tuned Models","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vengdeng/darex","path":"utils/evaluate_llms_utils.py","file_url":"https://github.com/vengdeng/darex/blob/HEAD/utils/evaluate_llms_utils.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2410.08196","paper":"/paper/mathcoder2-better-math-reasoning-from","title":"MathCoder2: Better Math Reasoning from Continued Pretraining on Model-translated Mathematical Code","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathllm/mathcoder2","path":"data_processing/mathematical_code/convert_to_text.py","file_url":"https://github.com/mathllm/mathcoder2/blob/HEAD/data_processing/mathematical_code/convert_to_text.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3adfdbf98eb1f8fe","mcp_get_code":{"code_sha256":"3adfdbf98eb1f8fe"}},{"arxiv_id":"2410.07985","paper":"/paper/omni-math-a-universal-olympiad-level","title":"Omni-MATH: A Universal Olympiad Level Mathematic Benchmark For Large Language Models","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kbsdjames/omni-math-rule","path":"evaluation/math_utils.py","file_url":"https://github.com/kbsdjames/omni-math-rule/blob/HEAD/evaluation/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1c1060f6e43d5dfa","mcp_get_code":{"code_sha256":"1c1060f6e43d5dfa"}},{"arxiv_id":"2409.02834","paper":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ecnu-icalk/educhat-math","path":"evaluation/utils.py","file_url":"https://github.com/ecnu-icalk/educhat-math/blob/HEAD/evaluation/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2409.00147","paper":"/paper/multimath-bridging-visual-and-mathematical","title":"MultiMath: Bridging Visual and Mathematical Reasoning for Large Language Models","date":"2024-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pengshuai-rin/multimath","path":"eval_mathverse/utils.py","file_url":"https://github.com/pengshuai-rin/multimath/blob/HEAD/eval_mathverse/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2409.00055","paper":"/paper/sorsa-singular-values-and-orthonormal","title":"SORSA: Singular Values and Orthonormal Regularized Singular Vectors Adaptation of Large Language Models","date":"2024-08-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Gunale0926/SORSA","path":"dataset.py","file_url":"https://github.com/Gunale0926/SORSA/blob/HEAD/dataset.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2408.10276","paper":"/paper/fedkim-adaptive-federated-knowledge-injection","title":"FEDKIM: Adaptive Federated Knowledge Injection into Medical Foundation Models","date":"2024-08-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XiaochenWang-PSU/FedKIM","path":"FedKIM/src/datasets/utils.py","file_url":"https://github.com/XiaochenWang-PSU/FedKIM/blob/HEAD/FedKIM/src/datasets/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"78e54f470017d052","mcp_get_code":{"code_sha256":"78e54f470017d052"}},{"arxiv_id":"2408.09227","paper":"/paper/fedmeki-a-benchmark-for-scaling-medical","title":"FEDMEKI: A Benchmark for Scaling Medical Foundation Models via Federated Knowledge Injection","date":"2024-08-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"psudslab/FEDMEKI","path":"FedMEKI/src/datasets/utils.py","file_url":"https://github.com/psudslab/FEDMEKI/blob/HEAD/FedMEKI/src/datasets/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"78e54f470017d052","mcp_get_code":{"code_sha256":"78e54f470017d052"}},{"arxiv_id":"2408.07930","paper":"/paper/mag-sql-multi-agent-generative-approach-with","title":"MAG-SQL: Multi-Agent Generative Approach with Soft Schema Linking and Iterative Sub-SQL Refinement for Text-to-SQL","date":"2024-08-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LancelotXWX/MAG-SQL","path":"main_scripts/bridge_content_encoder.py","file_url":"https://github.com/LancelotXWX/MAG-SQL/blob/HEAD/main_scripts/bridge_content_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f39fdc1ce6cffe08","mcp_get_code":{"code_sha256":"f39fdc1ce6cffe08"}},{"arxiv_id":"2408.04556","paper":"/paper/bias-aware-low-rank-adaptation-mitigating","title":"BA-LoRA: Bias-Alleviating Low-Rank Adaptation to Mitigate Catastrophic Inheritance in Large Language Models","date":"2024-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cyp-jlu-ai/ba-lora","path":"inference/gsm8k_inference.py","file_url":"https://github.com/cyp-jlu-ai/ba-lora/blob/HEAD/inference/gsm8k_inference.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2408.03092","paper":"/paper/extend-model-merging-from-fine-tuned-to-pre","title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","date":"2024-08-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yule-BUAA/MergeLLM","path":"utils/evaluate_llms_utils.py","file_url":"https://github.com/yule-BUAA/MergeLLM/blob/HEAD/utils/evaluate_llms_utils.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2407.21320","paper":"/paper/2407-21320","title":"MetaOpenFOAM: an LLM-based multi-agent framework for CFD","date":"2024-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"terry-cyx/metaopenfoam","path":"src/file_sta.py","file_url":"https://github.com/terry-cyx/metaopenfoam/blob/HEAD/src/file_sta.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2406.14024","paper":"/paper/the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kbsdjames/math-minos","path":"evaluation/eval_gsm8k.py","file_url":"https://github.com/kbsdjames/math-minos/blob/HEAD/evaluation/eval_gsm8k.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2406.12288","paper":"/paper/an-investigation-of-neuron-activation-as-a","title":"An Investigation of Neuron Activation as a Unified Lens to Explain Chain-of-Thought Eliciting Arithmetic Reasoning of LLMs","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dakingrai/ood-generalization-semantic-boundary-techniques","path":"evaluations/src/bridge_content_encoder.py","file_url":"https://github.com/dakingrai/ood-generalization-semantic-boundary-techniques/blob/HEAD/evaluations/src/bridge_content_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f39fdc1ce6cffe08","mcp_get_code":{"code_sha256":"f39fdc1ce6cffe08"}},{"arxiv_id":"2406.12084","paper":"/paper/when-reasoning-meets-information-aggregation","title":"When Reasoning Meets Information Aggregation: A Case Study with Sports Narratives","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yebowenhu/sportsgen","path":"utils/stats.py","file_url":"https://github.com/yebowenhu/sportsgen/blob/HEAD/utils/stats.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"316380478bf0bc96","mcp_get_code":{"code_sha256":"316380478bf0bc96"}},{"arxiv_id":"2406.05760","paper":"/paper/arabic-diacritics-in-the-wild-exploiting","title":"Arabic Diacritics in the Wild: Exploiting Opportunities for Improved Diacritization","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CAMeL-Lab/wild_diacritics","path":"code/wilddiacs_utils/modules/token_handler/token_handler.py","file_url":"https://github.com/CAMeL-Lab/wild_diacritics/blob/HEAD/code/wilddiacs_utils/modules/token_handler/token_handler.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-SA-4.0","inline_ok":false,"code_sha256_prefix":"c5cdf233d710d8fe","mcp_get_code":{"code_sha256":"c5cdf233d710d8fe"}},{"arxiv_id":"2405.15734","paper":"/paper/lm4lv-a-frozen-large-language-model-for-low","title":"LM4LV: A Frozen Large Language Model for Low-level Vision Tasks","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytetriper/lm4lv","path":"src/datasets/utils.py","file_url":"https://github.com/bytetriper/lm4lv/blob/HEAD/src/datasets/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"78e54f470017d052","mcp_get_code":{"code_sha256":"78e54f470017d052"}},{"arxiv_id":"2405.15179","paper":"/paper/vb-lora-extreme-parameter-efficient-fine","title":"VB-LoRA: Extreme Parameter Efficient Fine-Tuning with Vector Banks","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leo-yangli/VB-LoRA","path":"math_instruction_tuning/instruction_tuning_eval/gsm8k_eval.py","file_url":"https://github.com/leo-yangli/VB-LoRA/blob/HEAD/math_instruction_tuning/instruction_tuning_eval/gsm8k_eval.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2405.14365","paper":"/paper/jiuzhang3-0-efficiently-improving","title":"JiuZhang3.0: Efficiently Improving Mathematical Reasoning by Training Small Data Synthesis Models","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/jiuzhang3.0","path":"eval/math_utils.py","file_url":"https://github.com/rucaibox/jiuzhang3.0/blob/HEAD/eval/math_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1c1060f6e43d5dfa","mcp_get_code":{"code_sha256":"1c1060f6e43d5dfa"}},{"arxiv_id":"2405.14014","paper":"/paper/radarocc-robust-3d-occupancy-prediction-with","title":"RadarOcc: Robust 3D Occupancy Prediction with 4D Imaging Radar","date":"2024-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"toytiny/radarocc","path":"generate_4d_polar_percentil.py","file_url":"https://github.com/toytiny/radarocc/blob/HEAD/generate_4d_polar_percentil.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"06e19c1d21f057c6","mcp_get_code":{"code_sha256":"06e19c1d21f057c6"}},{"arxiv_id":"2405.06680","paper":"/paper/exploring-the-compositional-deficiency-of","title":"Exploring the Compositional Deficiency of Large Language Models in Mathematical Reasoning","date":"2024-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tongjingqi/MathTrap","path":"eval/eval_GSM8K_category.py","file_url":"https://github.com/tongjingqi/MathTrap/blob/HEAD/eval/eval_GSM8K_category.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2405.01719","paper":"/paper/inherent-trade-offs-between-diversity-and","title":"Inherent Trade-Offs between Diversity and Stability in Multi-Task Benchmarks","date":"2024-05-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"socialfoundations/benchbench","path":"benchbench/utils/base.py","file_url":"https://github.com/socialfoundations/benchbench/blob/HEAD/benchbench/utils/base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"27f819180313d7a4","mcp_get_code":{"code_sha256":"27f819180313d7a4"}},{"arxiv_id":"2404.05188","paper":"/paper/have-you-merged-my-model-on-the-robustness-of","title":"Have You Merged My Model? On The Robustness of Large Language Model IP Protection Methods Against Model Merging","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thuccslab/mergeguard","path":"eval_math.py","file_url":"https://github.com/thuccslab/mergeguard/blob/HEAD/eval_math.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2401.03201","paper":"/paper/3dmit-3d-multi-modal-instruction-tuning-for","title":"3DMIT: 3D Multi-modal Instruction Tuning for Scene Understanding","date":"2024-01-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"staymylove/3DMIT","path":"src/vg_eval_script.py","file_url":"https://github.com/staymylove/3DMIT/blob/HEAD/src/vg_eval_script.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"78e54f470017d052","mcp_get_code":{"code_sha256":"78e54f470017d052"}},{"arxiv_id":"2311.03099","paper":"/paper/language-models-are-super-mario-absorbing","title":"Language Models are Super Mario: Absorbing Abilities from Homologous Models as a Free Lunch","date":"2023-11-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yule-BUAA/MergeLM","path":"utils/evaluate_llms_utils.py","file_url":"https://github.com/yule-BUAA/MergeLM/blob/HEAD/utils/evaluate_llms_utils.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2310.13659","paper":"/paper/benchmarking-and-improving-text-to-sql","title":"Benchmarking and Improving Text-to-SQL Generation under Ambiguity","date":"2023-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"testzer0/AmbiQT","path":"src/utils/content.py","file_url":"https://github.com/testzer0/AmbiQT/blob/HEAD/src/utils/content.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f39fdc1ce6cffe08","mcp_get_code":{"code_sha256":"f39fdc1ce6cffe08"}},{"arxiv_id":"2310.08975","paper":"/paper/chatkbqa-a-generate-then-retrieve-framework","title":"ChatKBQA: A Generate-then-Retrieve Framework for Knowledge Base Question Answering with Fine-tuned Large Language Models","date":"2023-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lhrlab/chatkbqa","path":"eval_final.py","file_url":"https://github.com/lhrlab/chatkbqa/blob/HEAD/eval_final.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b2feca00ae40a160","mcp_get_code":{"code_sha256":"b2feca00ae40a160"}},{"arxiv_id":"2309.12284","paper":"/paper/metamath-bootstrap-your-own-mathematical","title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","date":"2023-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meta-math/MetaMath","path":"eval_gsm8k.py","file_url":"https://github.com/meta-math/MetaMath/blob/HEAD/eval_gsm8k.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2309.06275","paper":"/paper/2309-06275","title":"Re-Reading Improves Reasoning in Large Language Models","date":"2023-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tebmer/rereading-llm-reasoning","path":"utils/answer_parser.py","file_url":"https://github.com/tebmer/rereading-llm-reasoning/blob/HEAD/utils/answer_parser.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2309.05527","paper":"/paper/resimad-zero-shot-3d-domain-transfer-for","title":"ReSimAD: Zero-Shot 3D Domain Transfer for Autonomous Driving with Source Reconstruction and Target Simulation","date":"2023-09-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pjlab-adg/pcsim","path":"SurrogateMetricsCalculate/ComputeDNUC.py","file_url":"https://github.com/pjlab-adg/pcsim/blob/HEAD/SurrogateMetricsCalculate/ComputeDNUC.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2309.01339","paper":"/paper/unisa-unified-generative-framework-for-1","title":"UniSA: Unified Generative Framework for Sentiment Analysis","date":"2023-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dawn0815/UniSA","path":"src/generation.py","file_url":"https://github.com/dawn0815/UniSA/blob/HEAD/src/generation.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2308.12028","paper":"/paper/lkpnr-llm-and-kg-for-personalized-news","title":"LKPNR: LLM and KG for Personalized News Recommendation Framework","date":"2023-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xuan-zw/lkpnr","path":"NNR/MIND_corpus.py","file_url":"https://github.com/xuan-zw/lkpnr/blob/HEAD/NNR/MIND_corpus.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2306.08891","paper":"/paper/interleaving-pre-trained-language-models-and","title":"Interleaving Pre-Trained Language Models and Large Language Models for Zero-Shot NL2SQL Generation","date":"2023-06-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ruc-datalab/zeronl2sql","path":"src/utils/bridge_content_encoder.py","file_url":"https://github.com/ruc-datalab/zeronl2sql/blob/HEAD/src/utils/bridge_content_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f39fdc1ce6cffe08","mcp_get_code":{"code_sha256":"f39fdc1ce6cffe08"}},{"arxiv_id":"2305.11853","paper":"/paper/how-to-prompt-llms-for-text-to-sql-a-study-in","title":"How to Prompt LLMs for Text-to-SQL: A Study in Zero-shot, Single-domain, and Cross-domain Settings","date":"2023-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shuaichenchang/prompt-text-to-sql","path":"database_prompt_construction.py","file_url":"https://github.com/shuaichenchang/prompt-text-to-sql/blob/HEAD/database_prompt_construction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"af9bcffd70d6e3be","mcp_get_code":{"code_sha256":"af9bcffd70d6e3be"}},{"arxiv_id":"2302.08468","paper":"/paper/lever-learning-to-verify-language-to-code","title":"LEVER: Learning to Verify Language-to-Code Generation with Execution","date":"2023-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"niansong1996/lever","path":"execution/wtq_eval.py","file_url":"https://github.com/niansong1996/lever/blob/HEAD/execution/wtq_eval.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2212.09741","paper":"/paper/one-embedder-any-task-instruction-finetuned","title":"One Embedder, Any Task: Instruction-Finetuned Text Embeddings","date":"2022-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HKUNLP/instructor-embedding","path":"evaluation/prompt_retrieval/bridge_content_encoder.py","file_url":"https://github.com/HKUNLP/instructor-embedding/blob/HEAD/evaluation/prompt_retrieval/bridge_content_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f39fdc1ce6cffe08","mcp_get_code":{"code_sha256":"f39fdc1ce6cffe08"}},{"arxiv_id":"2201.06776","paper":"/paper/pruning-aware-sparse-regularization-for","title":"Pruning-aware Sparse Regularization for Network Pruning","date":"2022-01-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"casia-iva-lab/masksparsity","path":"netslim/prune.py","file_url":"https://github.com/casia-iva-lab/masksparsity/blob/HEAD/netslim/prune.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2110.04374","paper":"/paper/a-few-more-examples-may-be-worth-billions-of","title":"A Few More Examples May Be Worth Billions of Parameters","date":"2021-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuvalkirstain/lm-evaluation-harness","path":"lm_eval/utils.py","file_url":"https://github.com/yuvalkirstain/lm-evaluation-harness/blob/HEAD/lm_eval/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2109.05093","paper":"/paper/picard-parsing-incrementally-for-constrained","title":"PICARD: Parsing Incrementally for Constrained Auto-Regressive Decoding from Language Models","date":"2021-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ElementAI/picard","path":"seq2seq/utils/bridge_content_encoder.py","file_url":"https://github.com/ElementAI/picard/blob/HEAD/seq2seq/utils/bridge_content_encoder.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f39fdc1ce6cffe08","mcp_get_code":{"code_sha256":"f39fdc1ce6cffe08"}},{"arxiv_id":"2106.12144","paper":"/paper/nodepiece-compositional-and-parameter","title":"NodePiece: Compositional and Parameter-Efficient Representations of Large Knowledge Graphs","date":"2021-06-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AutoML-Research/KGBench","path":"kgbench/misc.py","file_url":"https://github.com/AutoML-Research/KGBench/blob/HEAD/kgbench/misc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3087bb1d9186ca95","mcp_get_code":{"code_sha256":"3087bb1d9186ca95"}},{"arxiv_id":"2105.07624","paper":"/paper/tat-qa-a-question-answering-benchmark-on-a","title":"TAT-QA: A Question Answering Benchmark on a Hybrid of Tabular and Textual Content in Finance","date":"2021-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NExTplusplus/TAT-QA","path":"tatqa_utils.py","file_url":"https://github.com/NExTplusplus/TAT-QA/blob/HEAD/tatqa_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1a0411f2ec690e26","mcp_get_code":{"code_sha256":"1a0411f2ec690e26"}},{"arxiv_id":"2009.11407","paper":"/paper/steering-a-historical-disease-forecasting","title":"Steering a Historical Disease Forecasting Model Under a Pandemic: Case of Flu and COVID-19","date":"2020-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AdityaLab/CALI-Net","path":"clustering_datasets.py","file_url":"https://github.com/AdityaLab/CALI-Net/blob/HEAD/clustering_datasets.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"2008.12813","paper":"/paper/hitter-hierarchical-transformers-for","title":"HittER: Hierarchical Transformers for Knowledge Graph Embeddings","date":"2020-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/HittER","path":"kge/misc.py","file_url":"https://github.com/microsoft/HittER/blob/HEAD/kge/misc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3087bb1d9186ca95","mcp_get_code":{"code_sha256":"3087bb1d9186ca95"}},{"arxiv_id":"2005.00558","paper":"/paper/pointer-constrained-text-generation-via","title":"POINTER: Constrained Progressive Text Generation via Insertion-based Generative Pre-training","date":"2020-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dreasysnail/POINTER","path":"keyword_extraction.py","file_url":"https://github.com/dreasysnail/POINTER/blob/HEAD/keyword_extraction.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"2003.05855","paper":"/paper/end-to-end-learning-local-multi-view","title":"End-to-End Learning Local Multi-view Descriptors for 3D Point Clouds","date":"2020-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"craigleili/3DLocalMultiViewDesc","path":"utils/io.py","file_url":"https://github.com/craigleili/3DLocalMultiViewDesc/blob/HEAD/utils/io.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"1906.02425","paper":"/paper/uncertainty-guided-continual-learning-with","title":"Uncertainty-guided Continual Learning with Bayesian Neural Networks","date":"2019-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SaynaEbrahimi/UCB","path":"src/utils.py","file_url":"https://github.com/SaynaEbrahimi/UCB/blob/HEAD/src/utils.py","status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"1807.06756","paper":"/paper/sysevr-a-framework-for-using-deep-learning-to","title":"SySeVR: A Framework for Using Deep Learning to Detect Software Vulnerabilities","date":"2018-07-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":2,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"0fe9cf6c00ef56ea","mcp_get_code":{"code_sha256":"0fe9cf6c00ef56ea"}},{"arxiv_id":"1705.09980","paper":"/paper/neural-semantic-parsing-by-character-based","title":"Neural Semantic Parsing by Character-based Translation: Experiments with Abstract Meaning Representations","date":"2017-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RikVN/AMR","path":"amr_utils.py","file_url":"https://github.com/RikVN/AMR/blob/HEAD/amr_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"1611.01734","paper":"/paper/deep-biaffine-attention-for-neural-dependency","title":"Deep Biaffine Attention for Neural Dependency Parsing","date":"2016-11-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JoesSattes/Thai-Biaffine-Dependency-Parsing","path":"preprocess.py","file_url":"https://github.com/JoesSattes/Thai-Biaffine-Dependency-Parsing/blob/HEAD/preprocess.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1889abe3c29f3531","mcp_get_code":{"code_sha256":"1889abe3c29f3531"}},{"arxiv_id":"1202.3665","paper":"/paper/emcee-the-mcmc-hammer","title":"emcee: The MCMC Hammer","date":null,"month_inferred_from_arxiv_id":"2012-02","title_source":"archive","repo":"s-ilic/ECLAIR","path":"ECLAIR_tools.py","file_url":"https://github.com/s-ilic/ECLAIR/blob/HEAD/ECLAIR_tools.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"openreview_XtIRCAEYoJ","paper":null,"title":"arXiv:openreview_XtIRCAEYoJ","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WayneTomas/Artemis","path":"val/refcoco_all/utils.py","file_url":"https://github.com/WayneTomas/Artemis/blob/HEAD/val/refcoco_all/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e18607dae38b30e2","mcp_get_code":{"code_sha256":"e18607dae38b30e2"}},{"arxiv_id":"aaai_17874","paper":null,"title":"arXiv:aaai_17874","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"hbaniecki/Pre-Surv-COVID-19","path":"utils.py","file_url":"https://github.com/hbaniecki/Pre-Surv-COVID-19/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f97aabe235dec048","mcp_get_code":{"code_sha256":"f97aabe235dec048"}},{"arxiv_id":"2025.findings-emnlp.638","paper":null,"title":"arXiv:2025.findings-emnlp.638","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"liyaooi/LongTableBench","path":"eval/reformat.py","file_url":"https://github.com/liyaooi/LongTableBench/blob/HEAD/eval/reformat.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e89c58b96c8bb4bc","mcp_get_code":{"code_sha256":"e89c58b96c8bb4bc"}},{"arxiv_id":"2020.emnlp-demos.22","paper":null,"title":"arXiv:2020.emnlp-demos.22","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"uma-pi1/kge","path":"kge/misc.py","file_url":"https://github.com/uma-pi1/kge/blob/HEAD/kge/misc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3087bb1d9186ca95","mcp_get_code":{"code_sha256":"3087bb1d9186ca95"}}]}