{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/read-jsonl","entry":"read_jsonl","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":136,"n_papers_ran":78,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":132,"n_samples_ran":66,"n_samples_fingerprinted":5,"n_places":152,"n_places_pointer_only":64,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":22,"ran_fixture":0,"ran":44,"unverified":66},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.13279","paper":"/paper/arxiv-2609-13279","title":"The MODA General Attribute Suite: A Four-Track Evaluation Benchmark for Fashion Attribute Extraction","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"hopit-ai/Moda_ner","path":"suite/catalog/score.py","file_url":"https://github.com/hopit-ai/Moda_ner/blob/HEAD/suite/catalog/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fa9634889181f1b8","mcp_get_code":{"code_sha256":"fa9634889181f1b8"}},{"arxiv_id":"2609.12584","paper":"/paper/arxiv-2609-12584","title":"Clustering-Based Balanced Sampling and Allocation with Data Parallelism for High-Performance Fine-Tuning","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"kaist-dmlab/CluSTER","path":"src/CluSTER/utils.py","file_url":"https://github.com/kaist-dmlab/CluSTER/blob/HEAD/src/CluSTER/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"764fe981599f1c88","mcp_get_code":{"code_sha256":"764fe981599f1c88"}},{"arxiv_id":"2609.02255","paper":"/paper/arxiv-2609-02255","title":"T2LSC-Bench: Benchmarking Localized Semantic Control in Text-to-Image Generation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"LLMSecResearch/T2LSC-Bench","path":"evaluation/metrics.py","file_url":"https://github.com/LLMSecResearch/T2LSC-Bench/blob/HEAD/evaluation/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"64690e0ae399636c","mcp_get_code":{"code_sha256":"64690e0ae399636c"}},{"arxiv_id":"2609.02133","paper":"/paper/arxiv-2609-02133","title":"EmoStance: Response-Side Affective-Orientation Control for Empathetic Response Generation via Emoji Weak Supervision","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"18277390221/EmoStance","path":"src/latent_stance_control/evaluate_text_quality.py","file_url":"https://github.com/18277390221/EmoStance/blob/HEAD/src/latent_stance_control/evaluate_text_quality.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"33d006261b75ac74","mcp_get_code":{"code_sha256":"33d006261b75ac74"}},{"arxiv_id":"2609.01081","paper":"/paper/arxiv-2609-01081","title":"StateSwap: Probing Support-Elimination Hidden States in Multiple-Choice Questions","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"Cha0Ga0/SWAPSTATE","path":"src/hs_swap/io.py","file_url":"https://github.com/Cha0Ga0/SWAPSTATE/blob/HEAD/src/hs_swap/io.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"725cbdee82cfaf24","mcp_get_code":{"code_sha256":"725cbdee82cfaf24"}},{"arxiv_id":"2609.00918","paper":"/paper/arxiv-2609-00918","title":"RPCBench: A Benchmark for Proactive Premise Critique in LLM-based Recommendation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"ZhongruChen/RPCBench","path":"src/step1/_step1_common.py","file_url":"https://github.com/ZhongruChen/RPCBench/blob/HEAD/src/step1/_step1_common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"569586b8eea9a84e","mcp_get_code":{"code_sha256":"569586b8eea9a84e"}},{"arxiv_id":"2609.00918","paper":"/paper/arxiv-2609-00918","title":"RPCBench: A Benchmark for Proactive Premise Critique in LLM-based Recommendation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"ZhongruChen/RPCBench","path":"src/step2_runner/_runner_common.py","file_url":"https://github.com/ZhongruChen/RPCBench/blob/HEAD/src/step2_runner/_runner_common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c52ed7dd243c3812","mcp_get_code":{"code_sha256":"c52ed7dd243c3812"}},{"arxiv_id":"2609.00275","paper":"/paper/arxiv-2609-00275","title":"The Irreversibility Budget: Fleet-Level Risk Accounting and Admission Control for Agent Operating Systems","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"mpi-dsg/irreversibility-budget","path":"sim/traces/plot.py","file_url":"https://github.com/mpi-dsg/irreversibility-budget/blob/HEAD/sim/traces/plot.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"953febbb771fc0e8","mcp_get_code":{"code_sha256":"953febbb771fc0e8"}},{"arxiv_id":"2608.23149","paper":"/paper/arxiv-2608-23149","title":"Language Chain in Alignment: Cross-lingual Ranking Preference Optimization","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"dltmddbs100/CRPO","path":"dataset/prepare_dataset.py","file_url":"https://github.com/dltmddbs100/CRPO/blob/HEAD/dataset/prepare_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"260bb4a4a14529db","mcp_get_code":{"code_sha256":"260bb4a4a14529db"}},{"arxiv_id":"2608.21365","paper":"/paper/arxiv-2608-21365","title":"KSE-Web: An Analysis of Hybrid Retrieval and LLM-Assisted Query Expansion for Low-Resource Khmer Semantic Search","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"back-kh/Khmer-Semantic-Search","path":"codes/evaluate_retrieval.py","file_url":"https://github.com/back-kh/Khmer-Semantic-Search/blob/HEAD/codes/evaluate_retrieval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4fc027c853affddb","mcp_get_code":{"code_sha256":"4fc027c853affddb"}},{"arxiv_id":"2608.21252","paper":"/paper/arxiv-2608-21252","title":"EnSI-RAG: Entity-Structure-Indexed Retrieval-Augmented Generation for Long-Document Question Answering","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"RamonMeng/EnSI-RAG","path":"ensi-rag-financebench-eval/ensi_financebench/qa_engine.py","file_url":"https://github.com/RamonMeng/EnSI-RAG/blob/HEAD/ensi-rag-financebench-eval/ensi_financebench/qa_engine.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c24e644ee0c0b239","mcp_get_code":{"code_sha256":"c24e644ee0c0b239"}},{"arxiv_id":"2607.28008","paper":"/paper/arxiv-2607-28008","title":"RepBench: A Benchmark for Representation Engineering","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"xlands/RepBench","path":"src/representation_cluster/representation_fit_pipeline.py","file_url":"https://github.com/xlands/RepBench/blob/HEAD/src/representation_cluster/representation_fit_pipeline.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7902fbf13b57a305","mcp_get_code":{"code_sha256":"7902fbf13b57a305"}},{"arxiv_id":"2607.14552","paper":"/paper/arxiv-2607-14552","title":"Answer-Conditioned Chains of Thought Degrade Verifiable-Reasoning Distillation in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"js-lee-AI/answer-leakage","path":"answer_leakage/corpus.py","file_url":"https://github.com/js-lee-AI/answer-leakage/blob/HEAD/answer_leakage/corpus.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"529acb6cf747d1fc","mcp_get_code":{"code_sha256":"529acb6cf747d1fc"}},{"arxiv_id":"2606.29920","paper":"/paper/arxiv-2606-29920","title":"Can LLM-as-a-Judge Reliably Verify Rubrics in Agentic Scenarios?","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"THU-KEG/RuVerBench","path":"code/run_judges/agenticcoding_runner.py","file_url":"https://github.com/THU-KEG/RuVerBench/blob/HEAD/code/run_judges/agenticcoding_runner.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"704044cedacb5584","mcp_get_code":{"code_sha256":"704044cedacb5584"}},{"arxiv_id":"2606.29920","paper":"/paper/arxiv-2606-29920","title":"Can LLM-as-a-Judge Reliably Verify Rubrics in Agentic Scenarios?","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"THU-KEG/RuVerBench","path":"code/validate_package.py","file_url":"https://github.com/THU-KEG/RuVerBench/blob/HEAD/code/validate_package.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"e7db7ab88e7a0bd2","mcp_get_code":{"code_sha256":"e7db7ab88e7a0bd2"}},{"arxiv_id":"2606.26979","paper":"/paper/arxiv-2606-26979","title":"How Much Static Structure Do Code Agents Need? A Study of Deterministic Anchoring","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"mathieu0905/Code-Anchor","path":"evaluation/eval_localization_codex.py","file_url":"https://github.com/mathieu0905/Code-Anchor/blob/HEAD/evaluation/eval_localization_codex.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"14b1625bd8d3d013","mcp_get_code":{"code_sha256":"14b1625bd8d3d013"}},{"arxiv_id":"2606.20661","paper":"/paper/arxiv-2606-20661","title":"From Knowing to Acting: Benchmarking Self-Awareness Capability of LLM Agents","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"AI-Santiago/KAware","path":"run_function_call_minimal.py","file_url":"https://github.com/AI-Santiago/KAware/blob/HEAD/run_function_call_minimal.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4afd53f9866aae14","mcp_get_code":{"code_sha256":"4afd53f9866aae14"}},{"arxiv_id":"2606.16535","paper":"/paper/arxiv-2606-16535","title":"Assessing Reliability of Symbol Detection in Concept Bottleneck Models","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Fuminides/cbm_sanity","path":"src/symbol_sanity/multihead.py","file_url":"https://github.com/Fuminides/cbm_sanity/blob/HEAD/src/symbol_sanity/multihead.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5b2dd5de02dc3211","mcp_get_code":{"code_sha256":"5b2dd5de02dc3211"}},{"arxiv_id":"2606.05647","paper":"/paper/arxiv-2606-05647","title":"Coding with \"Enemy\": Can Human Developers Detect AI Agent Sabotage?","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"CHATS-lab/coding-agent-safety-monitor","path":"utils/io.py","file_url":"https://github.com/CHATS-lab/coding-agent-safety-monitor/blob/HEAD/utils/io.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8add8881bedf4a28","mcp_get_code":{"code_sha256":"8add8881bedf4a28"}},{"arxiv_id":"2606.02578","paper":"/paper/arxiv-2606-02578","title":"Mitigating Perceptual Judgment Bias in Multimodal LLM-as-a-Judge via Perceptual Perturbation and Reward Modeling","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"kaist-cvml/perception-judge","path":"prepare-datasets/post_processing_merge_folders.py","file_url":"https://github.com/kaist-cvml/perception-judge/blob/HEAD/prepare-datasets/post_processing_merge_folders.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2d6a14cf6379542d","mcp_get_code":{"code_sha256":"2d6a14cf6379542d"}},{"arxiv_id":"2605.30794","paper":"/paper/arxiv-2605-30794","title":"MechVQA: Benchmarking and Enhancing Multimodal LLMs on Comprehensive Mechanical Drawing Understanding","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"xiaofengShi/MechVQA","path":"data_generation/generate_vqa_free_query.py","file_url":"https://github.com/xiaofengShi/MechVQA/blob/HEAD/data_generation/generate_vqa_free_query.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f73db1bba0fc60bf","mcp_get_code":{"code_sha256":"f73db1bba0fc60bf"}},{"arxiv_id":"2605.30794","paper":"/paper/arxiv-2605-30794","title":"MechVQA: Benchmarking and Enhancing Multimodal LLMs on Comprehensive Mechanical Drawing Understanding","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"xiaofengShi/MechVQA","path":"data_generation/check_generated_questions.py","file_url":"https://github.com/xiaofengShi/MechVQA/blob/HEAD/data_generation/check_generated_questions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0d8d1a6b7845a85d","mcp_get_code":{"code_sha256":"0d8d1a6b7845a85d"}},{"arxiv_id":"2605.28222","paper":"/paper/arxiv-2605-28222","title":"Analyzing Quality-Latency-Resource Trade-offs in a Technical Documentation RAG Assistant Using LoRA Adaptation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"EugPal/rag-lora-tradeoffs","path":"src/data_pipeline/build_kubernetes_qa_dataset.py","file_url":"https://github.com/EugPal/rag-lora-tradeoffs/blob/HEAD/src/data_pipeline/build_kubernetes_qa_dataset.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"65440b243bca4da1","mcp_get_code":{"code_sha256":"65440b243bca4da1"}},{"arxiv_id":"2605.25920","paper":"/paper/arxiv-2605-25920","title":"Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"AlexFanw/LegalSearch-R1","path":"user/calculate_metrics.py","file_url":"https://github.com/AlexFanw/LegalSearch-R1/blob/HEAD/user/calculate_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"27498a27e49ed413","mcp_get_code":{"code_sha256":"27498a27e49ed413"}},{"arxiv_id":"2605.12652","paper":"/paper/arxiv-2605-12652","title":"Multi-Rollout On-Policy Distillation via Peer Successes and Failures","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"viviable/mopd_code","path":"analysis/score_teacher_contexts.py","file_url":"https://github.com/viviable/mopd_code/blob/HEAD/analysis/score_teacher_contexts.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"903f3f3a478c7f63","mcp_get_code":{"code_sha256":"903f3f3a478c7f63"}},{"arxiv_id":"2605.12652","paper":"/paper/arxiv-2605-12652","title":"Multi-Rollout On-Policy Distillation via Peer Successes and Failures","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"viviable/mopd_code","path":"analysis/compute_teacher_signal_metrics.py","file_url":"https://github.com/viviable/mopd_code/blob/HEAD/analysis/compute_teacher_signal_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2b6d83de04b5e4ad","mcp_get_code":{"code_sha256":"2b6d83de04b5e4ad"}},{"arxiv_id":"2605.10195","paper":"/paper/arxiv-2605-10195","title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"PKU-SEC-Lab/SPEX","path":"bfs_async.py","file_url":"https://github.com/PKU-SEC-Lab/SPEX/blob/HEAD/bfs_async.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"951d55a786237f31","mcp_get_code":{"code_sha256":"951d55a786237f31"}},{"arxiv_id":"2605.04036","paper":"/paper/arxiv-2605-04036","title":"OpenSeeker-v2: Pushing the Limits of Search Agents with Informative and High-Difficulty Trajectories","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"PolarSeeker/OpenSeeker","path":"eval/generate_answer.py","file_url":"https://github.com/PolarSeeker/OpenSeeker/blob/HEAD/eval/generate_answer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"01c02f8e0d457fb1","mcp_get_code":{"code_sha256":"01c02f8e0d457fb1"}},{"arxiv_id":"2604.27998","paper":"/paper/arxiv-2604-27998","title":"Latent-GRPO: Group Relative Policy Optimization for Latent Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"DJC-GO-SOLO/Latent-SFT","path":"eval/eval_latent_model_hf_batch.py","file_url":"https://github.com/DJC-GO-SOLO/Latent-SFT/blob/HEAD/eval/eval_latent_model_hf_batch.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3d5343ae289698e8","mcp_get_code":{"code_sha256":"3d5343ae289698e8"}},{"arxiv_id":"2604.06834","paper":"/paper/arxiv-2604-06834","title":"On the Step Length Confounding in LLM Reasoning Data Selection","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"wangbing1416/ASLEC","path":"merge_cal_limo.py","file_url":"https://github.com/wangbing1416/ASLEC/blob/HEAD/merge_cal_limo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0a2bd131350902b8","mcp_get_code":{"code_sha256":"0a2bd131350902b8"}},{"arxiv_id":"2604.06834","paper":"/paper/arxiv-2604-06834","title":"On the Step Length Confounding in LLM Reasoning Data Selection","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"wangbing1416/ASLEC","path":"select_sft_data.py","file_url":"https://github.com/wangbing1416/ASLEC/blob/HEAD/select_sft_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8de9a2e11d7d03fa","mcp_get_code":{"code_sha256":"8de9a2e11d7d03fa"}},{"arxiv_id":"2604.06834","paper":"/paper/arxiv-2604-06834","title":"On the Step Length Confounding in LLM Reasoning Data Selection","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"wangbing1416/ASLEC","path":"select_sft_data_ours.py","file_url":"https://github.com/wangbing1416/ASLEC/blob/HEAD/select_sft_data_ours.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8fc83dfd8da40551","mcp_get_code":{"code_sha256":"8fc83dfd8da40551"}},{"arxiv_id":"2604.01705","paper":"/paper/arxiv-2604-01705","title":"Development and multi-center evaluation of domain-adapted speech recognition for human-AI teaming in real-world gastrointestinal endoscopy","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ku262/EndoASR","path":"eval/eval_acc.py","file_url":"https://github.com/ku262/EndoASR/blob/HEAD/eval/eval_acc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"41cce0cc5f4d5c38","mcp_get_code":{"code_sha256":"41cce0cc5f4d5c38"}},{"arxiv_id":"2603.11327","paper":"/paper/arxiv-2603-11327","title":"Meta-Reinforcement Learning with Self-Reflection for Agentic Search","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"tengxiao1/MR-Search","path":"meta-search/search/retrieval_server.py","file_url":"https://github.com/tengxiao1/MR-Search/blob/HEAD/meta-search/search/retrieval_server.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"02171abe8492fcfd","mcp_get_code":{"code_sha256":"02171abe8492fcfd"}},{"arxiv_id":"2603.11327","paper":"/paper/arxiv-2603-11327","title":"Meta-Reinforcement Learning with Self-Reflection for Agentic Search","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"tengxiao1/MR-Search","path":"meta-search/search/retrieval.py","file_url":"https://github.com/tengxiao1/MR-Search/blob/HEAD/meta-search/search/retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e50ece300cf79a17","mcp_get_code":{"code_sha256":"e50ece300cf79a17"}},{"arxiv_id":"2602.13576","paper":"/paper/arxiv-2602-13576","title":"Rubrics as an Attack Surface: Stealthy Preference Drift in LLM Judges","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ZDCSlab/Rubrics-as-an-Attack-Surface","path":"downstream_eval/eval/scores.py","file_url":"https://github.com/ZDCSlab/Rubrics-as-an-Attack-Surface/blob/HEAD/downstream_eval/eval/scores.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d97bc4cdc990894c","mcp_get_code":{"code_sha256":"d97bc4cdc990894c"}},{"arxiv_id":"2602.13576","paper":"/paper/arxiv-2602-13576","title":"Rubrics as an Attack Surface: Stealthy Preference Drift in LLM Judges","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ZDCSlab/Rubrics-as-an-Attack-Surface","path":"downstream_eval/eval/analyze.py","file_url":"https://github.com/ZDCSlab/Rubrics-as-an-Attack-Surface/blob/HEAD/downstream_eval/eval/analyze.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ffbf5293f2483abe","mcp_get_code":{"code_sha256":"ffbf5293f2483abe"}},{"arxiv_id":"2602.13576","paper":"/paper/arxiv-2602-13576","title":"Rubrics as an Attack Surface: Stealthy Preference Drift in LLM Judges","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ZDCSlab/Rubrics-as-an-Attack-Surface","path":"downstream_eval/eval/generate.py","file_url":"https://github.com/ZDCSlab/Rubrics-as-an-Attack-Surface/blob/HEAD/downstream_eval/eval/generate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f07107c7894990c1","mcp_get_code":{"code_sha256":"f07107c7894990c1"}},{"arxiv_id":"2602.13576","paper":"/paper/arxiv-2602-13576","title":"Rubrics as an Attack Surface: Stealthy Preference Drift in LLM Judges","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ZDCSlab/Rubrics-as-an-Attack-Surface","path":"downstream_eval/eval/select_best.py","file_url":"https://github.com/ZDCSlab/Rubrics-as-an-Attack-Surface/blob/HEAD/downstream_eval/eval/select_best.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6a94611fb33432fd","mcp_get_code":{"code_sha256":"6a94611fb33432fd"}},{"arxiv_id":"2602.11824","paper":"/paper/arxiv-2602-11824","title":"REVIS: Sparse Latent Steering to Mitigate Object Hallucination in Large Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"antgroup/Revis","path":"utils/chair.py","file_url":"https://github.com/antgroup/Revis/blob/HEAD/utils/chair.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"098ce1dbb8a719bf","mcp_get_code":{"code_sha256":"098ce1dbb8a719bf"}},{"arxiv_id":"2602.10863","paper":"/paper/arxiv-2602-10863","title":"ICA: Information-Aware Credit Assignment for Visually Grounded Long-Horizon Information-Seeking Agents","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"pc-inno/ICA_MM_deepsearch","path":"llm_judge.py","file_url":"https://github.com/pc-inno/ICA_MM_deepsearch/blob/HEAD/llm_judge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f45882c6806cb8d4","mcp_get_code":{"code_sha256":"f45882c6806cb8d4"}},{"arxiv_id":"2602.10021","paper":"/paper/arxiv-2602-10021","title":"Decoupled Reasoning with Implicit Fact Tokens (DRIFT): A Dual-Model Framework for Efficient Long-Context Inference","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"Lancelot-Xie/DRIFT","path":"src/drift/inference/eval_multi.py","file_url":"https://github.com/Lancelot-Xie/DRIFT/blob/HEAD/src/drift/inference/eval_multi.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c35fa39d6741b7b7","mcp_get_code":{"code_sha256":"c35fa39d6741b7b7"}},{"arxiv_id":"2602.02258","paper":"/paper/arxiv-2602-02258","title":"Alignment-Aware Model Adaptation via Feedback-Guided Optimization","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"facebookresearch/TruthRL","path":"data_utils/retrieval.py","file_url":"https://github.com/facebookresearch/TruthRL/blob/HEAD/data_utils/retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e50ece300cf79a17","mcp_get_code":{"code_sha256":"e50ece300cf79a17"}},{"arxiv_id":"2601.22047","paper":"/paper/arxiv-2601-22047","title":"On the Paradoxical Interference between Instruction-Following and Task Solving","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"kijlk/IF-Interference","path":"src/math_and_qa/evaluation.py","file_url":"https://github.com/kijlk/IF-Interference/blob/HEAD/src/math_and_qa/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9190ce603f740a88","mcp_get_code":{"code_sha256":"9190ce603f740a88"}},{"arxiv_id":"2601.14896","paper":"/paper/arxiv-2601-14896","title":"Language-Coupled Reinforcement Learning for Multilingual Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Cherry-qwq/LcRL-Open","path":"search_r1/search/retrieval.py","file_url":"https://github.com/Cherry-qwq/LcRL-Open/blob/HEAD/search_r1/search/retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e50ece300cf79a17","mcp_get_code":{"code_sha256":"e50ece300cf79a17"}},{"arxiv_id":"2601.09233","paper":"/paper/arxiv-2601-09233","title":"GIFT: Reconciling Post-Training Objectives via Finite-Temperature Gibbs Initialization","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"zzy1127/GIFT","path":"utils/data_utils.py","file_url":"https://github.com/zzy1127/GIFT/blob/HEAD/utils/data_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"82f198e67affc0bb","mcp_get_code":{"code_sha256":"82f198e67affc0bb"}},{"arxiv_id":"2510.13212","paper":"/paper/arxiv-2510-13212","title":"Towards Understanding Valuable Preference Data for Large Language Model Alignment","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"tmlr-group/TIF_LossDiff-IRM","path":"winrate_eval/single_score.py","file_url":"https://github.com/tmlr-group/TIF_LossDiff-IRM/blob/HEAD/winrate_eval/single_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a65ff6fa4d9d7cc7","mcp_get_code":{"code_sha256":"a65ff6fa4d9d7cc7"}},{"arxiv_id":"2510.13212","paper":"/paper/arxiv-2510-13212","title":"Towards Understanding Valuable Preference Data for Large Language Model Alignment","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"tmlr-group/TIF_LossDiff-IRM","path":"winrate_eval/single_score_local.py","file_url":"https://github.com/tmlr-group/TIF_LossDiff-IRM/blob/HEAD/winrate_eval/single_score_local.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"07adb23040d2275c","mcp_get_code":{"code_sha256":"07adb23040d2275c"}},{"arxiv_id":"2510.13212","paper":"/paper/arxiv-2510-13212","title":"Towards Understanding Valuable Preference Data for Large Language Model Alignment","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"tmlr-group/TIF_LossDiff-IRM","path":"analysis/data_select_by2_mid_k.py","file_url":"https://github.com/tmlr-group/TIF_LossDiff-IRM/blob/HEAD/analysis/data_select_by2_mid_k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"17f3dfa5c8d12d87","mcp_get_code":{"code_sha256":"17f3dfa5c8d12d87"}},{"arxiv_id":"2510.13212","paper":"/paper/arxiv-2510-13212","title":"Towards Understanding Valuable Preference Data for Large Language Model Alignment","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"tmlr-group/TIF_LossDiff-IRM","path":"analysis/data_select_mid_k.py","file_url":"https://github.com/tmlr-group/TIF_LossDiff-IRM/blob/HEAD/analysis/data_select_mid_k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e2a67da3a04cd3c7","mcp_get_code":{"code_sha256":"e2a67da3a04cd3c7"}},{"arxiv_id":"2509.25050","paper":"/paper/arxiv-2509-25050","title":"Advantage Weighted Matching: Aligning RL with Pretraining in Diffusion Models","date":"2025-09-29","month_inferred_from_arxiv_id":null,"title_source":"syntology","repo":"scxue/advantage_weighted_matching","path":"advantage_weighted_matching/dataset/merge_genevaltask.py","file_url":"https://github.com/scxue/advantage_weighted_matching/blob/HEAD/advantage_weighted_matching/dataset/merge_genevaltask.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"90a22cdf7d22703b","mcp_get_code":{"code_sha256":"90a22cdf7d22703b"}},{"arxiv_id":"2508.18847","paper":"/paper/arxiv-2508-18847","title":"ConfTuner: Training Large Language Models to Express Their Confidence Verbally","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"liushiliushi/ConfTuner","path":"src/llama_recipes/datasets2/gsm8k_dataset.py","file_url":"https://github.com/liushiliushi/ConfTuner/blob/HEAD/src/llama_recipes/datasets2/gsm8k_dataset.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"951d55a786237f31","mcp_get_code":{"code_sha256":"951d55a786237f31"}},{"arxiv_id":"2508.13037","paper":"/paper/arxiv-2508-13037","title":"Can Large Models Teach Student Models to Solve Mathematical Problems Like Human Beings? A Reasoning Distillation Method via Multi-LoRA Interaction","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"Xinhe-Li/LoRID","path":"src/utils/file_utils.py","file_url":"https://github.com/Xinhe-Li/LoRID/blob/HEAD/src/utils/file_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"19fb595831e42a78","mcp_get_code":{"code_sha256":"19fb595831e42a78"}},{"arxiv_id":"2507.02592","paper":"/paper/websailor-navigating-super-human-reasoning","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","date":"2025-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba-nlp/webagent","path":"WebAgent/NestBrowse/utils.py","file_url":"https://github.com/alibaba-nlp/webagent/blob/HEAD/WebAgent/NestBrowse/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ecb8ee265d233ddd","mcp_get_code":{"code_sha256":"ecb8ee265d233ddd"}},{"arxiv_id":"2506.10055","paper":"/paper/taskcraft-automated-generation-of-agentic","title":"TaskCraft: Automated Generation of Agentic Tasks","date":"2025-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"oppo-personalai/taskcraft","path":"taskcraft/utils.py","file_url":"https://github.com/oppo-personalai/taskcraft/blob/HEAD/taskcraft/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c5bf8e54f0728275","mcp_get_code":{"code_sha256":"c5bf8e54f0728275"}},{"arxiv_id":"2505.18822","paper":"/paper/adactrl-towards-adaptive-and-controllable","title":"AdaCtrl: Towards Adaptive and Controllable Reasoning via Difficulty-Aware Budgeting","date":"2025-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joeying1019/adactrl","path":"utils/data_utils.py","file_url":"https://github.com/joeying1019/adactrl/blob/HEAD/utils/data_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"82f198e67affc0bb","mcp_get_code":{"code_sha256":"82f198e67affc0bb"}},{"arxiv_id":"2505.17612","paper":"/paper/distilling-llm-agent-into-small-models-with","title":"Distilling LLM Agent into Small Models with Retrieval and Code Tools","date":"2025-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Nardien/agent-distillation","path":"search/retriever_server.py","file_url":"https://github.com/Nardien/agent-distillation/blob/HEAD/search/retriever_server.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"02171abe8492fcfd","mcp_get_code":{"code_sha256":"02171abe8492fcfd"}},{"arxiv_id":"2505.15210","paper":"/paper/deliberation-on-priors-trustworthy-reasoning","title":"Deliberation on Priors: Trustworthy Reasoning of Large Language Models on Knowledge Graphs","date":"2025-05-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mira-ai-lab/deliberation-on-priors","path":"reasoning/instantiation.py","file_url":"https://github.com/mira-ai-lab/deliberation-on-priors/blob/HEAD/reasoning/instantiation.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3d5343ae289698e8","mcp_get_code":{"code_sha256":"3d5343ae289698e8"}},{"arxiv_id":"2504.15895","paper":"/paper/dynamic-early-exit-in-reasoning-models","title":"Dynamic Early Exit in Reasoning Models","date":"2025-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iie-ycx/deer","path":"vllm-deer.py","file_url":"https://github.com/iie-ycx/deer/blob/HEAD/vllm-deer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c6485c4cde6c2322","mcp_get_code":{"code_sha256":"c6485c4cde6c2322"}},{"arxiv_id":"2504.08837","paper":"/paper/vl-rethinker-incentivizing-self-reflection-of","title":"VL-Rethinker: Incentivizing Self-Reflection of Vision-Language Models with Reinforcement Learning","date":"2025-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tiger-ai-lab/vl-rethinker","path":"openrlhf/trainer/evaluator.py","file_url":"https://github.com/tiger-ai-lab/vl-rethinker/blob/HEAD/openrlhf/trainer/evaluator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"059f7ea3562486c2","mcp_get_code":{"code_sha256":"059f7ea3562486c2"}},{"arxiv_id":"2503.18596","paper":"/paper/linkalign-scalable-schema-linking-for-real","title":"LinkAlign: Scalable Schema Linking for Real-World Large-Scale Multi-Database Text-to-SQL","date":"2025-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Satissss/LinkAlign","path":"preprocess.py","file_url":"https://github.com/Satissss/LinkAlign/blob/HEAD/preprocess.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a22422413f1c610b","mcp_get_code":{"code_sha256":"a22422413f1c610b"}},{"arxiv_id":"2503.15620","paper":"/paper/does-context-matter-contextualjudgebench-for","title":"Does Context Matter? ContextualJudgeBench for Evaluating LLM-based Judges in Contextual Settings","date":"2025-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforceairesearch/contextualjudgebench","path":"utils/utils.py","file_url":"https://github.com/salesforceairesearch/contextualjudgebench/blob/HEAD/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"c89138f7dd4645e9","mcp_get_code":{"code_sha256":"c89138f7dd4645e9"}},{"arxiv_id":"2502.13943","paper":"/paper/adaptivestep-automatically-dividing-reasoning","title":"AdaptiveStep: Automatically Dividing Reasoning Step through Model Confidence","date":"2025-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lux0926/asprm","path":"evaluation/code/TVD/tvd_lcb.py","file_url":"https://github.com/lux0926/asprm/blob/HEAD/evaluation/code/TVD/tvd_lcb.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"62569752807c6bc6","mcp_get_code":{"code_sha256":"62569752807c6bc6"}},{"arxiv_id":"2502.11196","paper":"/paper/how-do-llms-acquire-new-knowledge-a-knowledge","title":"How Do LLMs Acquire New Knowledge? A Knowledge Circuits Perspective on Continual Pre-Training","date":"2025-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zjunlp/dynamicknowledgecircuits","path":"utils.py","file_url":"https://github.com/zjunlp/dynamicknowledgecircuits/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bf43819905001d75","mcp_get_code":{"code_sha256":"bf43819905001d75"}},{"arxiv_id":"2502.05664","paper":"/paper/codesim-multi-agent-code-generation-and-1","title":"CODESIM: Multi-Agent Code Generation and Problem Solving through Simulation-Driven Planning and Debugging","date":"2025-02-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kagnlp/CodeGenerator","path":"src/datasets/convert-apps-xcode.py","file_url":"https://github.com/kagnlp/CodeGenerator/blob/HEAD/src/datasets/convert-apps-xcode.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"94c65e47ba360f89","mcp_get_code":{"code_sha256":"94c65e47ba360f89"}},{"arxiv_id":"2502.05173","paper":"/paper/videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wiselnn570/videorope","path":"eval/model_longvideobench_qwen2_vl.py","file_url":"https://github.com/wiselnn570/videorope/blob/HEAD/eval/model_longvideobench_qwen2_vl.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"15f446d6c10dc173","mcp_get_code":{"code_sha256":"15f446d6c10dc173"}},{"arxiv_id":"2412.16620","paper":"/paper/a-large-scale-empirical-study-on-fine-tuning","title":"A Large-scale Empirical Study on Fine-tuning Large Language Models for Unit Testing","date":null,"month_inferred_from_arxiv_id":"2024-12","title_source":"archive","repo":"iSEngLab/LLM4UT_Empirical","path":"Inference_Script/open_source.py","file_url":"https://github.com/iSEngLab/LLM4UT_Empirical/blob/HEAD/Inference_Script/open_source.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"02171abe8492fcfd","mcp_get_code":{"code_sha256":"02171abe8492fcfd"}},{"arxiv_id":"2412.11803","paper":"/paper/ualign-leveraging-uncertainty-estimations-for","title":"UAlign: Leveraging Uncertainty Estimations for Factuality Alignment on Large Language Models","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amourwaltz/ualign","path":"code/utils.py","file_url":"https://github.com/amourwaltz/ualign/blob/HEAD/code/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d5792c4c85064ef5","mcp_get_code":{"code_sha256":"d5792c4c85064ef5"}},{"arxiv_id":"2412.06846","paper":"/paper/classifier-free-guidance-in-llms-safety","title":"Classifier-free guidance in LLMs Safety","date":"2024-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rgsmirnov/cfg_safety_llm","path":"train_orpo.py","file_url":"https://github.com/rgsmirnov/cfg_safety_llm/blob/HEAD/train_orpo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"435e1bfd0feee04a","mcp_get_code":{"code_sha256":"435e1bfd0feee04a"}},{"arxiv_id":"2412.04905","paper":"/paper/demo-reframing-dialogue-interaction-with-fine","title":"DEMO: Reframing Dialogue Interaction with Fine-grained Element Modeling","date":"2024-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mozerwang/demo","path":"src/utils/functions.py","file_url":"https://github.com/mozerwang/demo/blob/HEAD/src/utils/functions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"aa2475b60b65226f","mcp_get_code":{"code_sha256":"aa2475b60b65226f"}},{"arxiv_id":"2412.02764","paper":"/paper/drawing-pandas-a-benchmark-for-llms-in","title":"Drawing Pandas: A Benchmark for LLMs in Generating Plotting Code","date":"2024-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jetbrains-research/pandasplotbench","path":"plotting_benchmark/vis_generator.py","file_url":"https://github.com/jetbrains-research/pandasplotbench/blob/HEAD/plotting_benchmark/vis_generator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e661693417b2d5b1","mcp_get_code":{"code_sha256":"e661693417b2d5b1"}},{"arxiv_id":"2412.01007","paper":"/paper/cornstack-high-quality-contrastive-data-for","title":"CoRNStack: High-Quality Contrastive Data for Better Code Retrieval and Reranking","date":"2024-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gangiswag/cornstack","path":"src/evaluations/eval_localization.py","file_url":"https://github.com/gangiswag/cornstack/blob/HEAD/src/evaluations/eval_localization.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4312a4596349487","mcp_get_code":{"code_sha256":"a4312a4596349487"}},{"arxiv_id":"2412.00535","paper":"/paper/fullstack-bench-evaluating-llms-as-full-stack","title":"FullStack Bench: Evaluating LLMs as Full Stack Coders","date":"2024-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/fullstackbench","path":"src/utils.py","file_url":"https://github.com/bytedance/fullstackbench/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ff3dacdd2cf428ba","mcp_get_code":{"code_sha256":"ff3dacdd2cf428ba"}},{"arxiv_id":"2411.17116","paper":"/paper/star-attention-efficient-llm-inference-over","title":"Star Attention: Efficient LLM Inference over Long Sequences","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NVIDIA/Star-Attention","path":"run_star_attn_inference.py","file_url":"https://github.com/NVIDIA/Star-Attention/blob/HEAD/run_star_attn_inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2a87c292d08948bc","mcp_get_code":{"code_sha256":"2a87c292d08948bc"}},{"arxiv_id":"2410.24198","paper":"/paper/selfcodealign-self-alignment-for-code","title":"SelfCodeAlign: Self-Alignment for Code Generation","date":"2024-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bigcode-project/starcoder2-self-align","path":"src/star_align/utils.py","file_url":"https://github.com/bigcode-project/starcoder2-self-align/blob/HEAD/src/star_align/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b0fdb2b203090d2e","mcp_get_code":{"code_sha256":"b0fdb2b203090d2e"}},{"arxiv_id":"2410.18077","paper":"/paper/alta-compiler-based-analysis-of-transformers","title":"ALTA: Compiler-Based Analysis of Transformers","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-deepmind/alta","path":"framework/common/io_utils.py","file_url":"https://github.com/google-deepmind/alta/blob/HEAD/framework/common/io_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bafd744639636cc4","mcp_get_code":{"code_sha256":"bafd744639636cc4"}},{"arxiv_id":"2410.15397","paper":"/paper/ipo-interpretable-prompt-optimization-for","title":"IPO: Interpretable Prompt Optimization for Vision-Language Models","date":"2024-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lmsdss/ipo","path":"optimization/eval_utils.py","file_url":"https://github.com/lmsdss/ipo/blob/HEAD/optimization/eval_utils.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4cf9d0d1ece601e","mcp_get_code":{"code_sha256":"a4cf9d0d1ece601e"}},{"arxiv_id":"2410.14309","paper":"/paper/logu-long-form-generation-with-uncertainty","title":"LoGU: Long-form Generation with Uncertainty Expressions","date":"2024-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rhyang2021/logu","path":"utils.py","file_url":"https://github.com/rhyang2021/logu/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a05eb9a0d2147891","mcp_get_code":{"code_sha256":"a05eb9a0d2147891"}},{"arxiv_id":"2410.13509","paper":"/paper/rag-ddr-optimizing-retrieval-augmented","title":"RAG-DDR: Optimizing Retrieval-Augmented Generation Using Differentiable Data Rewards","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openmatch/rag-ddr","path":"src/generator/gen_dpo_data.py","file_url":"https://github.com/openmatch/rag-ddr/blob/HEAD/src/generator/gen_dpo_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a22422413f1c610b","mcp_get_code":{"code_sha256":"a22422413f1c610b"}},{"arxiv_id":"2410.13509","paper":"/paper/rag-ddr-optimizing-retrieval-augmented","title":"RAG-DDR: Optimizing Retrieval-Augmented Generation Using Differentiable Data Rewards","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openmatch/rag-ddr","path":"src/generator/gen_llm_response.py","file_url":"https://github.com/openmatch/rag-ddr/blob/HEAD/src/generator/gen_llm_response.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fdd1249aa05ba5f3","mcp_get_code":{"code_sha256":"fdd1249aa05ba5f3"}},{"arxiv_id":"2410.13509","paper":"/paper/rag-ddr-optimizing-retrieval-augmented","title":"RAG-DDR: Optimizing Retrieval-Augmented Generation Using Differentiable Data Rewards","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openmatch/rag-ddr","path":"src/knowledgeRefinement/kr_inference.py","file_url":"https://github.com/openmatch/rag-ddr/blob/HEAD/src/knowledgeRefinement/kr_inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"90a22cdf7d22703b","mcp_get_code":{"code_sha256":"90a22cdf7d22703b"}},{"arxiv_id":"2410.12784","paper":"/paper/judgebench-a-benchmark-for-evaluating-llm","title":"JudgeBench: A Benchmark for Evaluating LLM-based Judges","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ScalerLab/JudgeBench","path":"utils/file_operations.py","file_url":"https://github.com/ScalerLab/JudgeBench/blob/HEAD/utils/file_operations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"195bbcd43396ba20","mcp_get_code":{"code_sha256":"195bbcd43396ba20"}},{"arxiv_id":"2410.10093","paper":"/paper/how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tengxiao1/GSIL","path":"gsil/combine.py","file_url":"https://github.com/tengxiao1/GSIL/blob/HEAD/gsil/combine.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ed3cbe8d977dbf1d","mcp_get_code":{"code_sha256":"ed3cbe8d977dbf1d"}},{"arxiv_id":"2410.09584","paper":"/paper/toward-general-instruction-following","title":"Toward General Instruction-Following Alignment for Retrieval-Augmented Generation","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongguanting/FollowRAG","path":"FollowRAG/utils/util.py","file_url":"https://github.com/dongguanting/FollowRAG/blob/HEAD/FollowRAG/utils/util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7bb90ae0136ac9dd","mcp_get_code":{"code_sha256":"7bb90ae0136ac9dd"}},{"arxiv_id":"2410.08081","paper":"/paper/packing-analysis-packing-is-more-appropriate","title":"Packing Analysis: Packing Is More Appropriate for Large Models or Datasets in Supervised Fine-tuning","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shuhewang1998/packing-analysis","path":"transfer_data.py","file_url":"https://github.com/shuhewang1998/packing-analysis/blob/HEAD/transfer_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"acd6bebe90eafbda","mcp_get_code":{"code_sha256":"acd6bebe90eafbda"}},{"arxiv_id":"2410.05695","paper":"/paper/unlocking-the-boundaries-of-thought-a","title":"Unlocking the Capabilities of Thought: A Reasoning Boundary Framework to Quantify and Optimize Chain-of-Thought","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LightChen233/reasoning-boundary","path":"utils/mm_tool.py","file_url":"https://github.com/LightChen233/reasoning-boundary/blob/HEAD/utils/mm_tool.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8f0ae0bd1b3d8a66","mcp_get_code":{"code_sha256":"8f0ae0bd1b3d8a66"}},{"arxiv_id":"2410.05695","paper":"/paper/unlocking-the-boundaries-of-thought-a","title":"Unlocking the Capabilities of Thought: A Reasoning Boundary Framework to Quantify and Optimize Chain-of-Thought","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lightchen233/reasoning-boundary","path":"draw_bound_text.py","file_url":"https://github.com/lightchen233/reasoning-boundary/blob/HEAD/draw_bound_text.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"86262555b2a8871d","mcp_get_code":{"code_sha256":"86262555b2a8871d"}},{"arxiv_id":"2410.02761","paper":"/paper/fakeshield-explainable-image-forgery","title":"FakeShield: Explainable Image Forgery Detection and Localization via Multi-modal Large Language Models","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhipeixu/fakeshield","path":"MFLM/cli_demo.py","file_url":"https://github.com/zhipeixu/fakeshield/blob/HEAD/MFLM/cli_demo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b1c0bb4ee3a78693","mcp_get_code":{"code_sha256":"b1c0bb4ee3a78693"}},{"arxiv_id":"2410.01692","paper":"/paper/u-shaped-and-inverted-u-scaling-behind","title":"U-shaped and Inverted-U Scaling behind Emergent Abilities of Large Language Models","date":"2024-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tony10101105/ExpEmergence","path":"evaluation/abstract_narrative_understanding/abstract_narrative_understanding_question_grouping.py","file_url":"https://github.com/tony10101105/ExpEmergence/blob/HEAD/evaluation/abstract_narrative_understanding/abstract_narrative_understanding_question_grouping.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"69f3426a987055c9","mcp_get_code":{"code_sha256":"69f3426a987055c9"}},{"arxiv_id":"2409.17892","paper":"/paper/emma-500-enhancing-massively-multilingual","title":"EMMA-500: Enhancing Massively Multilingual Adaptation of Large Language Models","date":"2024-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MaLA-LM/emma-500","path":"evaluation/Aya/evaluate.py","file_url":"https://github.com/MaLA-LM/emma-500/blob/HEAD/evaluation/Aya/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"47ad7e8829ffe863","mcp_get_code":{"code_sha256":"47ad7e8829ffe863"}},{"arxiv_id":"2409.05356","paper":"/paper/indicvoices-r-unlocking-a-massive","title":"IndicVoices-R: Unlocking a Massive Multilingual Multi-speaker Speech Corpus for Scaling Indian TTS","date":"2024-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ai4bharat/indicvoices-r","path":"VoiceCraft/inference.py","file_url":"https://github.com/ai4bharat/indicvoices-r/blob/HEAD/VoiceCraft/inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"c0e30f6224e1afda","mcp_get_code":{"code_sha256":"c0e30f6224e1afda"}},{"arxiv_id":"2408.15545","paper":"/paper/scilitllm-how-to-adapt-llms-for-scientific","title":"SciLitLLM: How to Adapt LLMs for Scientific Literature Understanding","date":"2024-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dptech-corp/Uni-SMART","path":"SciLitLLM/cpt/quality_control/llama_infer.py","file_url":"https://github.com/dptech-corp/Uni-SMART/blob/HEAD/SciLitLLM/cpt/quality_control/llama_infer.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"793cd3b000794433","mcp_get_code":{"code_sha256":"793cd3b000794433"}},{"arxiv_id":"2407.16695","paper":"/paper/stress-testing-long-context-language-models","title":"Stress-Testing Long-Context Language Models with Lifelong ICL and Task Haystack","date":"2024-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ink-usc/lifelong-icl","path":"dataloader/incremental_task.py","file_url":"https://github.com/ink-usc/lifelong-icl/blob/HEAD/dataloader/incremental_task.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9cfdd3d5a62805fa","mcp_get_code":{"code_sha256":"9cfdd3d5a62805fa"}},{"arxiv_id":"2407.12504","paper":"/paper/case2code-learning-inductive-reasoning-with","title":"Case2Code: Learning Inductive Reasoning with Synthetic Data","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"choosewhatulike/case2code","path":"api_call_util.py","file_url":"https://github.com/choosewhatulike/case2code/blob/HEAD/api_call_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"65c57366104cbd60","mcp_get_code":{"code_sha256":"65c57366104cbd60"}},{"arxiv_id":"2407.04965","paper":"/paper/beyond-perplexity-multi-dimensional-safety","title":"Beyond Perplexity: Multi-dimensional Safety Evaluation of LLM Compression","date":"2024-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhichaoxu-shufe/beyond-perplexity-compression-safety-eval","path":"src/dataset.py","file_url":"https://github.com/zhichaoxu-shufe/beyond-perplexity-compression-safety-eval/blob/HEAD/src/dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"39f2050e41061d06","mcp_get_code":{"code_sha256":"39f2050e41061d06"}},{"arxiv_id":"2406.18665","paper":"/paper/routellm-learning-to-route-llms-with","title":"RouteLLM: Learning to Route LLMs with Preference Data","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lm-sys/routellm","path":"routellm/evals/gsm8k/generate_responses.py","file_url":"https://github.com/lm-sys/routellm/blob/HEAD/routellm/evals/gsm8k/generate_responses.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d6397718adba76b0","mcp_get_code":{"code_sha256":"d6397718adba76b0"}},{"arxiv_id":"2406.16828","paper":"/paper/ragnarok-a-reusable-rag-framework-and","title":"Ragnarök: A Reusable RAG Framework and Baselines for TREC 2024 Retrieval-Augmented Generation Track","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"castorini/nuggetizer","path":"src/nuggetizer/cli/io.py","file_url":"https://github.com/castorini/nuggetizer/blob/HEAD/src/nuggetizer/cli/io.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"719779e74d1ac660","mcp_get_code":{"code_sha256":"719779e74d1ac660"}},{"arxiv_id":"2406.08100","paper":"/paper/multimodal-table-understanding","title":"Multimodal Table Understanding","date":"2024-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spursgozmy/table-llava","path":"llava/eval/generate_webpage_data_from_table.py","file_url":"https://github.com/spursgozmy/table-llava/blob/HEAD/llava/eval/generate_webpage_data_from_table.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7beeb01acc36919e","mcp_get_code":{"code_sha256":"7beeb01acc36919e"}},{"arxiv_id":"2406.06822","paper":"/paper/an-llm-assisted-easy-to-trigger-backdoor","title":"An LLM-Assisted Easy-to-Trigger Backdoor Attack on Code Completion Models: Injecting Disguised Vulnerabilities against Strong Detection","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"datasec-lab/codebreaker","path":"CodeGen/codegen1/benchmark/mtpb_exec.py","file_url":"https://github.com/datasec-lab/codebreaker/blob/HEAD/CodeGen/codegen1/benchmark/mtpb_exec.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b196c82bf4ad6d4d","mcp_get_code":{"code_sha256":"b196c82bf4ad6d4d"}},{"arxiv_id":"2406.04531","paper":"/paper/testeval-benchmarking-large-language-models","title":"TESTEVAL: Benchmarking Large Language Models for Test Case Generation","date":null,"month_inferred_from_arxiv_id":"2024-06","title_source":"archive","repo":"llm4softwaretesting/testeval","path":"data_utils.py","file_url":"https://github.com/llm4softwaretesting/testeval/blob/HEAD/data_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"60c584f55c9451fc","mcp_get_code":{"code_sha256":"60c584f55c9451fc"}},{"arxiv_id":"2405.19744","paper":"/paper/x-instruction-aligning-language-model-in-low","title":"X-Instruction: Aligning Language Model in Low-resource Languages with Self-curated Cross-lingual Instructions","date":"2024-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"znlp/x-instruction","path":"utils/chatgpt_generation.py","file_url":"https://github.com/znlp/x-instruction/blob/HEAD/utils/chatgpt_generation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6356690e9f0d26f0","mcp_get_code":{"code_sha256":"6356690e9f0d26f0"}},{"arxiv_id":"2405.18369","paper":"/paper/promptwizard-task-aware-agent-driven-prompt","title":"PromptWizard: Task-Aware Prompt Optimization Framework","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/promptwizard","path":"promptwizard/glue/paramlogger/file_utils.py","file_url":"https://github.com/microsoft/promptwizard/blob/HEAD/promptwizard/glue/paramlogger/file_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a18ff17da97103cd","mcp_get_code":{"code_sha256":"a18ff17da97103cd"}},{"arxiv_id":"2405.16802","paper":"/paper/autocv-empowering-reasoning-with-automated","title":"AutoPSV: Automated Process-Supervised Verifier","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rookie-joe/autocv","path":"utils/verfier_datasets.py","file_url":"https://github.com/rookie-joe/autocv/blob/HEAD/utils/verfier_datasets.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c409d17fa33773fa","mcp_get_code":{"code_sha256":"c409d17fa33773fa"}},{"arxiv_id":"2405.16473","paper":"/paper/m-3-cot-a-novel-benchmark-for-multi-domain","title":"M$^3$CoT: A Novel Benchmark for Multi-Domain Multi-step Multi-modal Chain-of-Thought","date":"2024-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LightChen233/M3CoT","path":"utils/data.py","file_url":"https://github.com/LightChen233/M3CoT/blob/HEAD/utils/data.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8f0ae0bd1b3d8a66","mcp_get_code":{"code_sha256":"8f0ae0bd1b3d8a66"}},{"arxiv_id":"2405.16405","paper":"/paper/intruding-with-words-towards-understanding","title":"Intruding with Words: Towards Understanding Graph Injection Attacks at the Text Level","date":"2024-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Leirunlin/Text-level-Graph-Attack","path":"LLM_utils.py","file_url":"https://github.com/Leirunlin/Text-level-Graph-Attack/blob/HEAD/LLM_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f6dadc26879ae08a","mcp_get_code":{"code_sha256":"f6dadc26879ae08a"}},{"arxiv_id":"2405.13382","paper":"/paper/vtg-llm-integrating-timestamp-knowledge-into","title":"VTG-LLM: Integrating Timestamp Knowledge into Video LLMs for Enhanced Video Temporal Grounding","date":"2024-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gyxxyg/vtg-llm","path":"utils/get_coco_format.py","file_url":"https://github.com/gyxxyg/vtg-llm/blob/HEAD/utils/get_coco_format.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8871a34f67963cf7","mcp_get_code":{"code_sha256":"8871a34f67963cf7"}},{"arxiv_id":"2405.11874","paper":"/paper/xfinder-robust-and-pinpoint-answer-extraction","title":"xFinder: Robust and Pinpoint Answer Extraction for Large Language Models","date":"2024-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IAAR-Shanghai/xFinder","path":"scripts/dataset_construction/create_benchmark_dataset.py","file_url":"https://github.com/IAAR-Shanghai/xFinder/blob/HEAD/scripts/dataset_construction/create_benchmark_dataset.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"352121e0de5ba153","mcp_get_code":{"code_sha256":"352121e0de5ba153"}},{"arxiv_id":"2404.10346","paper":"/paper/self-explore-to-avoid-the-pit-improving-the","title":"Self-Explore: Enhancing Mathematical Reasoning in Language Models with Fine-grained Rewards","date":"2024-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hbin0701/Self-Explore","path":"gen/get_rft_data.py","file_url":"https://github.com/hbin0701/Self-Explore/blob/HEAD/gen/get_rft_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c4783f2528e590ab","mcp_get_code":{"code_sha256":"c4783f2528e590ab"}},{"arxiv_id":"2404.04042","paper":"/paper/teaching-llama-a-new-language-through-cross","title":"Teaching Llama a New Language Through Cross-Lingual Knowledge Transfer","date":"2024-04-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tartunlp/llammas","path":"scripts/chat_data/instructions_to_chat.py","file_url":"https://github.com/tartunlp/llammas/blob/HEAD/scripts/chat_data/instructions_to_chat.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9d32f043536de9b2","mcp_get_code":{"code_sha256":"9d32f043536de9b2"}},{"arxiv_id":"2404.01288","paper":"/paper/large-language-models-are-capable-of-offering","title":"Large Language Models are Capable of Offering Cognitive Reappraisal, if Guided","date":"2024-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"honglizhan/resort_cognitive_reappraisal","path":"utils.py","file_url":"https://github.com/honglizhan/resort_cognitive_reappraisal/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ca3d4025d7524b34","mcp_get_code":{"code_sha256":"ca3d4025d7524b34"}},{"arxiv_id":"2403.18252","paper":"/paper/beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/visual-table","path":"llava/eval/generate_webpage_data_from_table.py","file_url":"https://github.com/lavi-lab/visual-table/blob/HEAD/llava/eval/generate_webpage_data_from_table.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7beeb01acc36919e","mcp_get_code":{"code_sha256":"7beeb01acc36919e"}},{"arxiv_id":"2403.18120","paper":"/paper/don-t-trust-verify-grounding-llm-quantitative","title":"Don't Trust: Verify -- Grounding LLM Quantitative Reasoning with Autoformalization","date":"2024-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jinpz/dtv","path":"formalize_proof.py","file_url":"https://github.com/jinpz/dtv/blob/HEAD/formalize_proof.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1eee05718afec3bf","mcp_get_code":{"code_sha256":"1eee05718afec3bf"}},{"arxiv_id":"2403.13372","paper":"/paper/llamafactory-unified-efficient-fine-tuning-of","title":"LlamaFactory: Unified Efficient Fine-Tuning of 100+ Language Models","date":"2024-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Rcrossmeister/RLQG","path":"model/converter.py","file_url":"https://github.com/Rcrossmeister/RLQG/blob/HEAD/model/converter.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2a7c632d17841b02","mcp_get_code":{"code_sha256":"2a7c632d17841b02"}},{"arxiv_id":"2402.18540","paper":"/paper/keeping-llms-aligned-after-fine-tuning-the","title":"Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vfleaking/PTST","path":"gpt-api/data_utils/prep_data.py","file_url":"https://github.com/vfleaking/PTST/blob/HEAD/gpt-api/data_utils/prep_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"64c1b0186895ea4a","mcp_get_code":{"code_sha256":"64c1b0186895ea4a"}},{"arxiv_id":"2402.14007","paper":"/paper/can-watermarks-survive-translation-on-the","title":"Can Watermarks Survive Translation? On the Cross-lingual Consistency of Text Watermark for Large Language Models","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zwhe99/x-sir","path":"utils.py","file_url":"https://github.com/zwhe99/x-sir/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8f17a72bd2c9a658","mcp_get_code":{"code_sha256":"8f17a72bd2c9a658"}},{"arxiv_id":"2402.13963","paper":"/paper/towards-building-multilingual-language-model","title":"Towards Building Multilingual Language Model for Medicine","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"magic-ai4med/mmedlm","path":"inference/inference.py","file_url":"https://github.com/magic-ai4med/mmedlm/blob/HEAD/inference/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"7aa433b5d2a47cd0","mcp_get_code":{"code_sha256":"7aa433b5d2a47cd0"}},{"arxiv_id":"2402.13125","paper":"/paper/treeeval-benchmark-free-evaluation-of-large","title":"TreeEval: Benchmark-Free Evaluation of Large Language Models through Tree Planning","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ashura5/treeeval","path":"score.py","file_url":"https://github.com/ashura5/treeeval/blob/HEAD/score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1c38f26b19003c2a","mcp_get_code":{"code_sha256":"1c38f26b19003c2a"}},{"arxiv_id":"2402.11845","paper":"/paper/modularized-networks-for-few-shot-hateful","title":"Modularized Networks for Few-shot Hateful Meme Detection","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"social-ai-studio/mod_hate","path":"src/gen_dataset.py","file_url":"https://github.com/social-ai-studio/mod_hate/blob/HEAD/src/gen_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"93b9095b31acfe8e","mcp_get_code":{"code_sha256":"93b9095b31acfe8e"}},{"arxiv_id":"2402.03720","paper":"/paper/similarity-based-neighbor-selection-for-graph","title":"Similarity-based Neighbor Selection for Graph LLMs","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ruili33/sns","path":"utils.py","file_url":"https://github.com/ruili33/sns/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9e0ab4c54364054f","mcp_get_code":{"code_sha256":"9e0ab4c54364054f"}},{"arxiv_id":"2401.13478","paper":"/paper/scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wusiwei0410/scimmir","path":"classify_training_data.py","file_url":"https://github.com/wusiwei0410/scimmir/blob/HEAD/classify_training_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fcc053d17ca95e43","mcp_get_code":{"code_sha256":"fcc053d17ca95e43"}},{"arxiv_id":"2401.10768","paper":"/paper/mitigating-hallucinations-of-large-language","title":"Knowledge Verification to Nip Hallucination in the Bud","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fanqiwan/KCA","path":"data_generation/inconsistency_processing.py","file_url":"https://github.com/fanqiwan/KCA/blob/HEAD/data_generation/inconsistency_processing.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"02171abe8492fcfd","mcp_get_code":{"code_sha256":"02171abe8492fcfd"}},{"arxiv_id":"2401.06081","paper":"/paper/improving-large-language-models-via-fine","title":"Improving Large Language Models via Fine-grained Reinforcement Learning with Minimum Editing Constraint","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/rlmec","path":"evaluate/Math/utils.py","file_url":"https://github.com/rucaibox/rlmec/blob/HEAD/evaluate/Math/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"83a3a266232cc7d0","mcp_get_code":{"code_sha256":"83a3a266232cc7d0"}},{"arxiv_id":"2312.12832","paper":"/paper/turning-dust-into-gold-distilling-complex","title":"Turning Dust into Gold: Distilling Complex Reasoning Capabilities from LLMs by Leveraging Negative Data","date":"2023-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yiwei98/TDG","path":"code/dataset.py","file_url":"https://github.com/Yiwei98/TDG/blob/HEAD/code/dataset.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"951d55a786237f31","mcp_get_code":{"code_sha256":"951d55a786237f31"}},{"arxiv_id":"2311.16452","paper":"/paper/can-generalist-foundation-models-outcompete","title":"Can Generalist Foundation Models Outcompete Special-Purpose Tuning? Case Study in Medicine","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/promptbase","path":"src/promptbase/drop/drop.py","file_url":"https://github.com/microsoft/promptbase/blob/HEAD/src/promptbase/drop/drop.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4be24c2f6782adf1","mcp_get_code":{"code_sha256":"4be24c2f6782adf1"}},{"arxiv_id":"2311.16079","paper":"/paper/meditron-70b-scaling-medical-pretraining-for","title":"MEDITRON-70B: Scaling Medical Pretraining for Large Language Models","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"epfllm/meditron","path":"gap-replay/guidelines/clean.py","file_url":"https://github.com/epfllm/meditron/blob/HEAD/gap-replay/guidelines/clean.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"63ecbb0137ec2cf0","mcp_get_code":{"code_sha256":"63ecbb0137ec2cf0"}},{"arxiv_id":"2310.14799","paper":"/paper/cross-lingual-prompting-improving-zero-shot","title":"Cross-lingual Prompting: Improving Zero-shot Chain-of-Thought Reasoning across Languages","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lightchen233/cross-lingual-prompting","path":"utils/clsp_metric.py","file_url":"https://github.com/lightchen233/cross-lingual-prompting/blob/HEAD/utils/clsp_metric.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"19802b15b25696b1","mcp_get_code":{"code_sha256":"19802b15b25696b1"}},{"arxiv_id":"2310.11511","paper":"/paper/self-rag-learning-to-retrieve-generate-and","title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fate-ubw/raglab","path":"raglab/rag/train_alg/finetune.py","file_url":"https://github.com/fate-ubw/raglab/blob/HEAD/raglab/rag/train_alg/finetune.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6242dd60b264c96f","mcp_get_code":{"code_sha256":"6242dd60b264c96f"}},{"arxiv_id":"2310.10158","paper":"/paper/character-llm-a-trainable-agent-for-role","title":"Character-LLM: A Trainable Agent for Role-Playing","date":"2023-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"choosewhatulike/trainable-agents","path":"io_utils.py","file_url":"https://github.com/choosewhatulike/trainable-agents/blob/HEAD/io_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3151efbd64278742","mcp_get_code":{"code_sha256":"3151efbd64278742"}},{"arxiv_id":"2310.07736","paper":"/paper/observatory-characterizing-embeddings-of","title":"Observatory: Characterizing Embeddings of Relational Tables","date":"2023-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"superctj/observatory","path":"observatory/models/example_doduo_entity_embeddings.py","file_url":"https://github.com/superctj/observatory/blob/HEAD/observatory/models/example_doduo_entity_embeddings.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bdc5cba97b33c234","mcp_get_code":{"code_sha256":"bdc5cba97b33c234"}},{"arxiv_id":"2310.07177","paper":"/paper/online-speculative-decoding","title":"Online Speculative Decoding","date":"2023-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuxiaoxuanpku/osd","path":"distill/data.py","file_url":"https://github.com/liuxiaoxuanpku/osd/blob/HEAD/distill/data.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"951d55a786237f31","mcp_get_code":{"code_sha256":"951d55a786237f31"}},{"arxiv_id":"2310.06147","paper":"/paper/reinforcement-learning-in-the-era-of-llms","title":"Reinforcement Learning in the Era of LLMs: What is Essential? What is needed? An RL Perspective on RLHF, Prompting, and Beyond","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vanderschaarlab/prompt-oirl","path":"llama_exps/llama_step1_gen_offline.py","file_url":"https://github.com/vanderschaarlab/prompt-oirl/blob/HEAD/llama_exps/llama_step1_gen_offline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"951d55a786237f31","mcp_get_code":{"code_sha256":"951d55a786237f31"}},{"arxiv_id":"2309.06553","paper":"/paper/offline-prompt-evaluation-and-optimization","title":"Query-Dependent Prompt Evaluation and Optimization with Offline Inverse RL","date":"2023-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"951d55a786237f31","mcp_get_code":{"code_sha256":"951d55a786237f31"}},{"arxiv_id":"2309.03409","paper":"/paper/large-language-models-as-optimizers","title":"Large Language Models as Optimizers","date":"2023-09-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-deepmind/opro","path":"opro/evaluation/eval_utils.py","file_url":"https://github.com/google-deepmind/opro/blob/HEAD/opro/evaluation/eval_utils.py","status":"ran_draft_wrong","verification_level":2,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4cf9d0d1ece601e","mcp_get_code":{"code_sha256":"a4cf9d0d1ece601e"}},{"arxiv_id":"2308.03279","paper":"/paper/universalner-targeted-distillation-from-large","title":"UniversalNER: Targeted Distillation from Large Language Models for Open Named Entity Recognition","date":"2023-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"universal-ner/universal-ner","path":"src/train/fastchat/eval/generate_webpage_data_from_table.py","file_url":"https://github.com/universal-ner/universal-ner/blob/HEAD/src/train/fastchat/eval/generate_webpage_data_from_table.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7beeb01acc36919e","mcp_get_code":{"code_sha256":"7beeb01acc36919e"}},{"arxiv_id":"2307.03067","paper":"/paper/deeponto-a-python-package-for-ontology","title":"DeepOnto: A Python Package for Ontology Engineering with Deep Learning","date":"2023-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KRR-Oxford/DeepOnto","path":"src/deeponto/utils/file_utils.py","file_url":"https://github.com/KRR-Oxford/DeepOnto/blob/HEAD/src/deeponto/utils/file_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ad437c9f60ec4a6","mcp_get_code":{"code_sha256":"6ad437c9f60ec4a6"}},{"arxiv_id":"2306.00887","paper":"/paper/openpi-c-a-better-benchmark-and-stronger","title":"OpenPI-C: A Better Benchmark and Stronger Baseline for Open-Vocabulary State Tracking","date":"2023-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shirley-wu/openpi-c","path":"training/gen_ans_to_list.py","file_url":"https://github.com/shirley-wu/openpi-c/blob/HEAD/training/gen_ans_to_list.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"abd1ef77eb0194bb","mcp_get_code":{"code_sha256":"abd1ef77eb0194bb"}},{"arxiv_id":"2304.10453","paper":"/paper/phoenix-democratizing-chatgpt-across","title":"Phoenix: Democratizing ChatGPT across Languages","date":"2023-04-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/llmzoo","path":"llmzoo/eval/compute_metric_all.py","file_url":"https://github.com/freedomintelligence/llmzoo/blob/HEAD/llmzoo/eval/compute_metric_all.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7beeb01acc36919e","mcp_get_code":{"code_sha256":"7beeb01acc36919e"}},{"arxiv_id":"2304.01904","paper":"/paper/refiner-reasoning-feedback-on-intermediate","title":"REFINER: Reasoning Feedback on Intermediate Representations","date":"2023-04-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"debjitpaul/refiner","path":"data_preprocessing/utils.py","file_url":"https://github.com/debjitpaul/refiner/blob/HEAD/data_preprocessing/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a573277f152a72e1","mcp_get_code":{"code_sha256":"a573277f152a72e1"}},{"arxiv_id":"2211.09374","paper":"/paper/execution-based-evaluation-for-data-science","title":"Execution-based Evaluation for Data Science Code Generation Models","date":"2022-11-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jun-jie-huang/exeds","path":"preprocess/preprocess.py","file_url":"https://github.com/jun-jie-huang/exeds/blob/HEAD/preprocess/preprocess.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d8eb246a783ad9f4","mcp_get_code":{"code_sha256":"d8eb246a783ad9f4"}},{"arxiv_id":"2205.12647","paper":"/paper/overcoming-catastrophic-forgetting-in-zero","title":"Overcoming Catastrophic Forgetting in Zero-Shot Cross-Lingual Generation","date":"2022-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google-research/prompt-tuning","path":"prompt_tuning/recycling/collect_recycling_results.py","file_url":"https://github.com/google-research/prompt-tuning/blob/HEAD/prompt_tuning/recycling/collect_recycling_results.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cb6ca8f7f505ca27","mcp_get_code":{"code_sha256":"cb6ca8f7f505ca27"}},{"arxiv_id":"2205.11601","paper":"/paper/challenges-in-measuring-bias-via-open-ended","title":"Challenges in Measuring Bias via Open-Ended Language Generation","date":"2022-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"feyzaakyurek/bias-textgen","path":"jsonl2txt.py","file_url":"https://github.com/feyzaakyurek/bias-textgen/blob/HEAD/jsonl2txt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ae46698901774aff","mcp_get_code":{"code_sha256":"ae46698901774aff"}},{"arxiv_id":"2204.04288","paper":"/paper/unsupervised-uncertainty-measures-of","title":"Unsupervised Uncertainty Measures of Automatic Speech Recognition for Non-intrusive Speech Intelligibility Prediction","date":"2022-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"claritychallenge/clarity","path":"clarity/utils/file_io.py","file_url":"https://github.com/claritychallenge/clarity/blob/HEAD/clarity/utils/file_io.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1e9637f47001093b","mcp_get_code":{"code_sha256":"1e9637f47001093b"}},{"arxiv_id":"2109.08535","paper":"/paper/simple-entity-centric-questions-challenge","title":"Simple Entity-Centric Questions Challenge Dense Retrievers","date":"2021-09-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/EntityQuestions","path":"utils/ion.py","file_url":"https://github.com/princeton-nlp/EntityQuestions/blob/HEAD/utils/ion.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9b6d4e8dece9a1b5","mcp_get_code":{"code_sha256":"9b6d4e8dece9a1b5"}},{"arxiv_id":"2108.00573","paper":"/paper/musique-multi-hop-questions-via-single-hop","title":"MuSiQue: Multihop Questions via Single-hop Question Composition","date":"2021-08-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stonybrooknlp/musique","path":"evaluate_v1.0.py","file_url":"https://github.com/stonybrooknlp/musique/blob/HEAD/evaluate_v1.0.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"a0c3c47032ca66fd","mcp_get_code":{"code_sha256":"a0c3c47032ca66fd"}},{"arxiv_id":"2104.06486","paper":"/paper/ms2-multi-document-summarization-of-medical","title":"MS2: Multi-Document Summarization of Medical Studies","date":"2021-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/ms2","path":"ms2/models/pubmed_tagger.py","file_url":"https://github.com/allenai/ms2/blob/HEAD/ms2/models/pubmed_tagger.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a5d2696481d31b52","mcp_get_code":{"code_sha256":"a5d2696481d31b52"}},{"arxiv_id":"2009.10795","paper":"/paper/dataset-cartography-mapping-and-diagnosing","title":"Dataset Cartography: Mapping and Diagnosing Datasets with Training Dynamics","date":"2020-09-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/cartography","path":"cartography/data_utils.py","file_url":"https://github.com/allenai/cartography/blob/HEAD/cartography/data_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d4036398be7d8fc9","mcp_get_code":{"code_sha256":"d4036398be7d8fc9"}},{"arxiv_id":"1810.05334","paper":"/paper/indosum-a-new-benchmark-dataset-for","title":"IndoSum: A New Benchmark Dataset for Indonesian Text Summarization","date":"2018-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kata-ai/indosum","path":"prep_outliers.py","file_url":"https://github.com/kata-ai/indosum/blob/HEAD/prep_outliers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cf5f87ab64fc2a27","mcp_get_code":{"code_sha256":"cf5f87ab64fc2a27"}},{"arxiv_id":"openreview_9aU4vrHPKD","paper":null,"title":"arXiv:openreview_9aU4vrHPKD","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Octobrist/CoPE","path":"src/custom/llm-as-the-judge.py","file_url":"https://github.com/Octobrist/CoPE/blob/HEAD/src/custom/llm-as-the-judge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"979fd42ef2935b6c","mcp_get_code":{"code_sha256":"979fd42ef2935b6c"}},{"arxiv_id":"2025.findings-acl.318","paper":null,"title":"arXiv:2025.findings-acl.318","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"SCUNLP/BAR","path":"bar/utils/util_box.py","file_url":"https://github.com/SCUNLP/BAR/blob/HEAD/bar/utils/util_box.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d6fd3bcc16758dfb","mcp_get_code":{"code_sha256":"d6fd3bcc16758dfb"}},{"arxiv_id":"2025.findings-acl.301","paper":null,"title":"arXiv:2025.findings-acl.301","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenBMB/ConsJudge","path":"src/RAG_train/infer_acc.py","file_url":"https://github.com/OpenBMB/ConsJudge/blob/HEAD/src/RAG_train/infer_acc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"aa2475b60b65226f","mcp_get_code":{"code_sha256":"aa2475b60b65226f"}},{"arxiv_id":"2025.emnlp-main.1337","paper":null,"title":"arXiv:2025.emnlp-main.1337","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"RazvanDu/CopySpec","path":"find_best_k/best_k_cs.py","file_url":"https://github.com/RazvanDu/CopySpec/blob/HEAD/find_best_k/best_k_cs.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"04f99874105cf826","mcp_get_code":{"code_sha256":"04f99874105cf826"}},{"arxiv_id":"2022.emnlp-demos.25","paper":null,"title":"arXiv:2022.emnlp-demos.25","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LogiTorch/logitorch","path":"src/logitorch/datasets/utils.py","file_url":"https://github.com/LogiTorch/logitorch/blob/HEAD/src/logitorch/datasets/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d93296a39d8edd4a","mcp_get_code":{"code_sha256":"d93296a39d8edd4a"}}]}