{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/calculate-metrics","entry":"calculate_metrics","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":61,"n_papers_ran":29,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":62,"n_samples_ran":26,"n_samples_fingerprinted":5,"n_places":65,"n_places_pointer_only":32,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":8,"ran":14,"unverified":36},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2607.06796","paper":"/paper/arxiv-2607-06796","title":"Enhancing deep learning models for time series classification via knowledge distillation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"MSD-IRIMAS/KD-4-TSC","path":"KD4TSC/utils.py","file_url":"https://github.com/MSD-IRIMAS/KD-4-TSC/blob/HEAD/KD4TSC/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"802e2c8e966af2b1","mcp_get_code":{"code_sha256":"802e2c8e966af2b1"}},{"arxiv_id":"2606.16617","paper":"/paper/arxiv-2606-16617","title":"Sycophancy as Material Failure under Pushback Loading A Multi-Axis Characterization Across Three Loading Cases and up to Seventeen Material Charges","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"JiseungHong/SYCON-Bench","path":"debate_setting/evaluate_base_ToF.py","file_url":"https://github.com/JiseungHong/SYCON-Bench/blob/HEAD/debate_setting/evaluate_base_ToF.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0a10e54cb945833c","mcp_get_code":{"code_sha256":"0a10e54cb945833c"}},{"arxiv_id":"2606.14900","paper":"/paper/arxiv-2606-14900","title":"GRASP: Gradient-Aligned Sequential Parameter Transfer for Memory-Efficient Multi-Source Learning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Sekeh-Lab/grasp-multisource-transfer","path":"experiments/grasp/run_grasp_experiment.py","file_url":"https://github.com/Sekeh-Lab/grasp-multisource-transfer/blob/HEAD/experiments/grasp/run_grasp_experiment.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e950bb9bbf851272","mcp_get_code":{"code_sha256":"e950bb9bbf851272"}},{"arxiv_id":"2605.18565","paper":"/paper/arxiv-2605-18565","title":"MINTEVAL: Evaluating Memory under Multi-Target Interference in Long-Horizon Agent Systems","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"amy-hyunji/MINTEval","path":"src/mem_alpha/memalpha/llm_agent/metrics.py","file_url":"https://github.com/amy-hyunji/MINTEval/blob/HEAD/src/mem_alpha/memalpha/llm_agent/metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c9e9a810f3e053fb","mcp_get_code":{"code_sha256":"c9e9a810f3e053fb"}},{"arxiv_id":"2604.08284","paper":"/paper/arxiv-2604-08284","title":"Distributed Multi-Layer Editing for Rule-Level Knowledge in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Pepper66/DMLE","path":"cal_avg.py","file_url":"https://github.com/Pepper66/DMLE/blob/HEAD/cal_avg.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fcc0593f8dc28f9d","mcp_get_code":{"code_sha256":"fcc0593f8dc28f9d"}},{"arxiv_id":"2604.08131","paper":"/paper/arxiv-2604-08131","title":"Graph Neural Networks for Misinformation Detection: Performance-Efficiency Trade-offs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"mkrzywda/gnn-misinformation-tradeoffs","path":"GNN-modified.py","file_url":"https://github.com/mkrzywda/gnn-misinformation-tradeoffs/blob/HEAD/GNN-modified.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"0d211a14cf22e405","mcp_get_code":{"code_sha256":"0d211a14cf22e405"}},{"arxiv_id":"2604.08131","paper":"/paper/arxiv-2604-08131","title":"Graph Neural Networks for Misinformation Detection: Performance-Efficiency Trade-offs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"mkrzywda/gnn-misinformation-tradeoffs","path":"baselines.py","file_url":"https://github.com/mkrzywda/gnn-misinformation-tradeoffs/blob/HEAD/baselines.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"0fd5d745f5999db6","mcp_get_code":{"code_sha256":"0fd5d745f5999db6"}},{"arxiv_id":"2603.28387","paper":"/paper/arxiv-2603-28387","title":"Prompts Without Evidence: How Neuroimaging Mentions Shift Clinical Vision-Language Model Predictions","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"long21wt/scaffold-effect","path":"src/f1_eval.py","file_url":"https://github.com/long21wt/scaffold-effect/blob/HEAD/src/f1_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f068fcc1e4587d2f","mcp_get_code":{"code_sha256":"f068fcc1e4587d2f"}},{"arxiv_id":"2603.28387","paper":"/paper/arxiv-2603-28387","title":"Prompts Without Evidence: How Neuroimaging Mentions Shift Clinical Vision-Language Model Predictions","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"long21wt/scaffold-effect","path":"src/f1_eval_oasis.py","file_url":"https://github.com/long21wt/scaffold-effect/blob/HEAD/src/f1_eval_oasis.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a035fc9a992b738f","mcp_get_code":{"code_sha256":"a035fc9a992b738f"}},{"arxiv_id":"2603.21298","paper":"/paper/arxiv-2603-21298","title":"More Than Sum of Its Parts: Deciphering Intent Shifts in Multimodal Hate Speech Detection","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Sayur1n/H-VLI","path":"evaluator.py","file_url":"https://github.com/Sayur1n/H-VLI/blob/HEAD/evaluator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"cf1eb64ff5a7a55a","mcp_get_code":{"code_sha256":"cf1eb64ff5a7a55a"}},{"arxiv_id":"2602.13139","paper":"/paper/arxiv-2602-13139","title":"OpenLID-v3: Improving the Precision of Closely Related Language Identification -An Experience Report","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ltgoslo/slide","path":"src/evaluate.py","file_url":"https://github.com/ltgoslo/slide/blob/HEAD/src/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8651b2739b0c4b96","mcp_get_code":{"code_sha256":"8651b2739b0c4b96"}},{"arxiv_id":"2602.11824","paper":"/paper/arxiv-2602-11824","title":"REVIS: Sparse Latent Steering to Mitigate Object Hallucination in Large Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"antgroup/Revis","path":"utils/mmvet_judge_only.py","file_url":"https://github.com/antgroup/Revis/blob/HEAD/utils/mmvet_judge_only.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"64c97831bbe858d3","mcp_get_code":{"code_sha256":"64c97831bbe858d3"}},{"arxiv_id":"2602.10458","paper":"/paper/arxiv-2602-10458","title":"Found-RL: foundation model-enhanced reinforcement learning for autonomous driving","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"ys-qu/found-rl","path":"clean_wandb_data.py","file_url":"https://github.com/ys-qu/found-rl/blob/HEAD/clean_wandb_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"cad6e72086af2eb4","mcp_get_code":{"code_sha256":"cad6e72086af2eb4"}},{"arxiv_id":"2602.08913","paper":"/paper/arxiv-2602-08913","title":"GEMSS: A Variational Method for Discovering Multiple Sparse Solutions in Classification and Regression Problems","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"kat-er-ina/gemss_testing","path":"algorithm_comparison/src/evaluation.py","file_url":"https://github.com/kat-er-ina/gemss_testing/blob/HEAD/algorithm_comparison/src/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5e372b1774b9c227","mcp_get_code":{"code_sha256":"5e372b1774b9c227"}},{"arxiv_id":"2602.07063","paper":"/paper/arxiv-2602-07063","title":"Video-based Music Generation","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"serkansulun/trailer-genre-classification","path":"classification/src/utils_classify.py","file_url":"https://github.com/serkansulun/trailer-genre-classification/blob/HEAD/classification/src/utils_classify.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"4d37c5189d2765e3","mcp_get_code":{"code_sha256":"4d37c5189d2765e3"}},{"arxiv_id":"2602.03895","paper":"/paper/arxiv-2602-03895","title":"Benchmarking Bias Mitigation Toward Fairness Without Harm from Vision to LVLMs","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"osu-srml/NH-Fair","path":"src/release_benchmark/methods/vlm/clip_fairer.py","file_url":"https://github.com/osu-srml/NH-Fair/blob/HEAD/src/release_benchmark/methods/vlm/clip_fairer.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b3ee986ec34da06e","mcp_get_code":{"code_sha256":"b3ee986ec34da06e"}},{"arxiv_id":"2602.01433","paper":"/paper/arxiv-2602-01433","title":"DCD: Decomposition-based Causal Discovery from Autocorrelated and Non-Stationary Temporal Data","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"noname2122/DCD-TMLR-2025","path":"experiments/run_dynotears_baseline.py","file_url":"https://github.com/noname2122/DCD-TMLR-2025/blob/HEAD/experiments/run_dynotears_baseline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9500b6a629b1e792","mcp_get_code":{"code_sha256":"9500b6a629b1e792"}},{"arxiv_id":"2601.08867","paper":"/paper/arxiv-2601-08867","title":"R 2 BD: A Reconstruction-Based Method for Generalizable and Efficient Detection of Fake Images","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"QingyuLiu/RRBD","path":"utils.py","file_url":"https://github.com/QingyuLiu/RRBD/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d0600278ff835cb1","mcp_get_code":{"code_sha256":"d0600278ff835cb1"}},{"arxiv_id":"2601.03699","paper":"/paper/arxiv-2601-03699","title":"ICLR 2026 Workshop: Principled Design for Trustworthy AI REDBENCH: A UNIVERSAL DATASET FOR COMPRE-HENSIVE RED TEAMING OF LARGE LANGUAGE MOD-ELS","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"knoveleng/redeval","path":"redeval/score.py","file_url":"https://github.com/knoveleng/redeval/blob/HEAD/redeval/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"036e0114f72d9a8c","mcp_get_code":{"code_sha256":"036e0114f72d9a8c"}},{"arxiv_id":"2510.12119","paper":"/paper/arxiv-2510-12119","title":"ImageSentinel: Protecting Visual Datasets from Unauthorized Retrieval-Augmented Image Generation","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"luo-ziyuan/ImageSentinel","path":"evaluate_similarities.py","file_url":"https://github.com/luo-ziyuan/ImageSentinel/blob/HEAD/evaluate_similarities.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c18eab3d2bf63ca5","mcp_get_code":{"code_sha256":"c18eab3d2bf63ca5"}},{"arxiv_id":"2506.14035","paper":"/paper/simpledoc-multi-modal-document-understanding","title":"SimpleDoc: Multi-Modal Document Understanding with Dual-Cue Page Retrieval and Iterative Refinement","date":"2025-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ag2ai/SimpleDoc","path":"evaluation/analyze_multiple_runs.py","file_url":"https://github.com/ag2ai/SimpleDoc/blob/HEAD/evaluation/analyze_multiple_runs.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"36d783e0916c23cb","mcp_get_code":{"code_sha256":"36d783e0916c23cb"}},{"arxiv_id":"2506.05945","paper":"/paper/on-efficient-estimation-of-distributional","title":"On Efficient Estimation of Distributional Treatment Effects under Covariate-Adaptive Randomization","date":"2025-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CyberAgentAILab/dte_car","path":"simulation.py","file_url":"https://github.com/CyberAgentAILab/dte_car/blob/HEAD/simulation.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"534c7b1b790218d9","mcp_get_code":{"code_sha256":"534c7b1b790218d9"}},{"arxiv_id":"2505.20840","paper":null,"title":"arXiv:2505.20840","date":null,"month_inferred_from_arxiv_id":"2025-05","title_source":null,"repo":"dooho00/agg-buffer","path":"evaluate/metrics.py","file_url":"https://github.com/dooho00/agg-buffer/blob/HEAD/evaluate/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b7d98c7e91eb3959","mcp_get_code":{"code_sha256":"b7d98c7e91eb3959"}},{"arxiv_id":"2505.16994","paper":"/paper/text-r-2-text-ec-towards-large-recommender","title":"$\\text{R}^2\\text{ec}$: Towards Large Recommender Models with Reasoning","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YRYangang/RRec","path":"trainers/utils.py","file_url":"https://github.com/YRYangang/RRec/blob/HEAD/trainers/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0bc8c7e05959c420","mcp_get_code":{"code_sha256":"0bc8c7e05959c420"}},{"arxiv_id":"2503.14492","paper":"/paper/cosmos-transfer1-conditional-world-generation","title":"Cosmos-Transfer1: Conditional World Generation with Adaptive Multimodal Control","date":"2025-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nv-tlabs/cosmos-drive-dreams","path":"cosmos-drive-dreams-toolkits/train_light_model.py","file_url":"https://github.com/nv-tlabs/cosmos-drive-dreams/blob/HEAD/cosmos-drive-dreams-toolkits/train_light_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55d080a2eba4d6b7","mcp_get_code":{"code_sha256":"55d080a2eba4d6b7"}},{"arxiv_id":"2412.10535","paper":"/paper/on-adversarial-robustness-and-out-of","title":"On Adversarial Robustness and Out-of-Distribution Robustness of Large Language Models","date":"2024-12-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jordantab/llm-robustness-experiment","path":"utilities/calculate_pb_baseline.py","file_url":"https://github.com/jordantab/llm-robustness-experiment/blob/HEAD/utilities/calculate_pb_baseline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"54807fe1fc74ab13","mcp_get_code":{"code_sha256":"54807fe1fc74ab13"}},{"arxiv_id":"2412.10535","paper":"/paper/on-adversarial-robustness-and-out-of","title":"On Adversarial Robustness and Out-of-Distribution Robustness of Large Language Models","date":"2024-12-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jordantab/llm-robustness-experiment","path":"utilities/parse.py","file_url":"https://github.com/jordantab/llm-robustness-experiment/blob/HEAD/utilities/parse.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c8436ea20f798b9a","mcp_get_code":{"code_sha256":"c8436ea20f798b9a"}},{"arxiv_id":"2411.04421","paper":"/paper/variational-low-rank-adaptation-using-ivon","title":"Variational Low-Rank Adaptation Using IVON","date":"2024-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"team-approx-bayes/ivon-lora","path":"utils.py","file_url":"https://github.com/team-approx-bayes/ivon-lora/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"436f4c309d8668cf","mcp_get_code":{"code_sha256":"436f4c309d8668cf"}},{"arxiv_id":"2410.11055","paper":"/paper/varying-shades-of-wrong-aligning-llms-with","title":"Varying Shades of Wrong: Aligning LLMs with Wrong Answers Only","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yaojh18/Varying-Shades-of-Wrong","path":"preference_optimization/evaluate.py","file_url":"https://github.com/yaojh18/Varying-Shades-of-Wrong/blob/HEAD/preference_optimization/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"29adfabf77b2b461","mcp_get_code":{"code_sha256":"29adfabf77b2b461"}},{"arxiv_id":"2410.03396","paper":"/paper/graphcroc-cross-correlation-autoencoder-for","title":"GraphCroc: Cross-Correlation Autoencoder for Graph Structural Reconstruction","date":"2024-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sjduan/graphcroc","path":"IMDB_B/reconstructor.py","file_url":"https://github.com/sjduan/graphcroc/blob/HEAD/IMDB_B/reconstructor.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"033a97925e132d68","mcp_get_code":{"code_sha256":"033a97925e132d68"}},{"arxiv_id":"2409.14083","paper":"/paper/surf-teaching-large-vision-language-models-to","title":"SURf: Teaching Large Vision-Language Models to Selectively Utilize Retrieved Information","date":"2024-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GasolSun36/SURf","path":"eval/eval_pope.py","file_url":"https://github.com/GasolSun36/SURf/blob/HEAD/eval/eval_pope.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"31fd3eb13a63be04","mcp_get_code":{"code_sha256":"31fd3eb13a63be04"}},{"arxiv_id":"2408.07888","paper":"/paper/fine-tuning-large-language-models-with-human","title":"Evaluating Fine-Tuning Efficiency of Human-Inspired Learning Strategies in Medical Question Answering","date":"2024-08-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Oxford-AI-for-Society/human-learning-strategies","path":"training/inference/inference.py","file_url":"https://github.com/Oxford-AI-for-Society/human-learning-strategies/blob/HEAD/training/inference/inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"57123d949a952552","mcp_get_code":{"code_sha256":"57123d949a952552"}},{"arxiv_id":"2408.06276","paper":"/paper/review-driven-personalized-preference","title":"Review-driven Personalized Preference Reasoning with Large Language Models for Recommendation","date":"2024-08-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jieyong99/exp3rt","path":"test_result_inspect.py","file_url":"https://github.com/jieyong99/exp3rt/blob/HEAD/test_result_inspect.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d46e7f90d226d0e6","mcp_get_code":{"code_sha256":"d46e7f90d226d0e6"}},{"arxiv_id":"2406.20015","paper":"/paper/toolbehonest-a-multi-level-hallucination","title":"ToolBeHonest: A Multi-level Hallucination Diagnostic Benchmark for Tool-Augmented Large Language Models","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"toolbehonest/toolbehonest","path":"utils/calculate_metrics.py","file_url":"https://github.com/toolbehonest/toolbehonest/blob/HEAD/utils/calculate_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"391471f665ef5fbb","mcp_get_code":{"code_sha256":"391471f665ef5fbb"}},{"arxiv_id":"2405.16871","paper":"/paper/multi-behavior-generative-recommendation","title":"Multi-Behavior Generative Recommendation","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anananan116/MBGen","path":"trainer/evaluation.py","file_url":"https://github.com/anananan116/MBGen/blob/HEAD/trainer/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bdbd3b67520ae048","mcp_get_code":{"code_sha256":"bdbd3b67520ae048"}},{"arxiv_id":"2405.11618","paper":"/paper/transcriptomics-guided-slide-representation","title":"Transcriptomics-guided Slide Representation Learning in Computational Pathology","date":"2024-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mahmoodlab/tangle","path":"run_linear_probing.py","file_url":"https://github.com/mahmoodlab/tangle/blob/HEAD/run_linear_probing.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"dc22df19cd5e4ef3","mcp_get_code":{"code_sha256":"dc22df19cd5e4ef3"}},{"arxiv_id":"2403.18624","paper":"/paper/vulnerability-detection-with-code-language","title":"Vulnerability Detection with Code Language Models: How Far Are We?","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dlvuldet/primevul","path":"os_expr/run_ft.py","file_url":"https://github.com/dlvuldet/primevul/blob/HEAD/os_expr/run_ft.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7559f2b5dd244067","mcp_get_code":{"code_sha256":"7559f2b5dd244067"}},{"arxiv_id":"2403.17359","paper":"/paper/chain-of-action-faithful-and-multimodal","title":"Chain-of-Action: Faithful and Multimodal Question Answering through Large Language Models","date":"2024-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MAGICS-LAB/Chain-of-Actions","path":"chain-of-search.py","file_url":"https://github.com/MAGICS-LAB/Chain-of-Actions/blob/HEAD/chain-of-search.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"483518008187ff8f","mcp_get_code":{"code_sha256":"483518008187ff8f"}},{"arxiv_id":"2403.02839","paper":"/paper/an-empirical-study-of-llm-as-a-judge-for-llm","title":"An Empirical Study of LLM-as-a-Judge for LLM Evaluation: Fine-tuned Judge Model is not a General Substitute for GPT-4","date":"2024-03-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huihuichyan/unlimitedjudge","path":"src/build_dataset.py","file_url":"https://github.com/huihuichyan/unlimitedjudge/blob/HEAD/src/build_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"acacb923208da24f","mcp_get_code":{"code_sha256":"acacb923208da24f"}},{"arxiv_id":"2402.17423","paper":"/paper/reinforced-in-context-black-box-optimization","title":"Reinforced In-Context Black-Box Optimization","date":"2024-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"songlei00/ribbo","path":"algorithms/utils.py","file_url":"https://github.com/songlei00/ribbo/blob/HEAD/algorithms/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cac93e27abd92689","mcp_get_code":{"code_sha256":"cac93e27abd92689"}},{"arxiv_id":"2401.04861","paper":"/paper/ctnerf-cross-time-transformer-for-dynamic","title":"CTNeRF: Cross-Time Transformer for Dynamic Neural Radiance Field from Monocular Video","date":"2024-01-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xingy038/ctnerf","path":"utils/evaluation.py","file_url":"https://github.com/xingy038/ctnerf/blob/HEAD/utils/evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d42c9ddd0cc88b91","mcp_get_code":{"code_sha256":"d42c9ddd0cc88b91"}},{"arxiv_id":"2311.01449","paper":"/paper/topicgpt-a-prompt-based-topic-modeling","title":"TopicGPT: A Prompt-based Topic Modeling Framework","date":"2023-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chtmp223/topicgpt","path":"topicgpt_python/utils.py","file_url":"https://github.com/chtmp223/topicgpt/blob/HEAD/topicgpt_python/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0d8395537f4bf2da","mcp_get_code":{"code_sha256":"0d8395537f4bf2da"}},{"arxiv_id":"2310.18685","paper":"/paper/when-reviewers-lock-horn-finding-disagreement","title":"When Reviewers Lock Horn: Finding Disagreement in Scientific Peer Reviews","date":"2023-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sandeep82945/contradiction-in-peer-review","path":"src/training_scratch3.py","file_url":"https://github.com/sandeep82945/contradiction-in-peer-review/blob/HEAD/src/training_scratch3.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8379fdb8173a8157","mcp_get_code":{"code_sha256":"8379fdb8173a8157"}},{"arxiv_id":"2310.15047","paper":"/paper/meta-out-of-context-learning-in-neural","title":"Implicit meta-learning may lead language models to trust more reliable sources","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"krasheninnikov/internalization","path":"src/fewshot.py","file_url":"https://github.com/krasheninnikov/internalization/blob/HEAD/src/fewshot.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"27effc0e19a1a9df","mcp_get_code":{"code_sha256":"27effc0e19a1a9df"}},{"arxiv_id":"2309.13345","paper":"/paper/bamboo-a-comprehensive-benchmark-for","title":"BAMBOO: A Comprehensive Benchmark for Evaluating Long Text Modeling Capacities of Large Language Models","date":"2023-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/bamboo","path":"evaluate.py","file_url":"https://github.com/rucaibox/bamboo/blob/HEAD/evaluate.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a8b0c9b324725da9","mcp_get_code":{"code_sha256":"a8b0c9b324725da9"}},{"arxiv_id":"2306.13531","paper":"/paper/wbcatt-a-white-blood-cell-dataset-annotated-1","title":"WBCAtt: A White Blood Cell Dataset Annotated with Detailed Morphological Attributes","date":"2023-06-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apple2373/wbcatt","path":"submission/traineval.py","file_url":"https://github.com/apple2373/wbcatt/blob/HEAD/submission/traineval.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6bc4993019f86d1a","mcp_get_code":{"code_sha256":"6bc4993019f86d1a"}},{"arxiv_id":"2306.07280","paper":"/paper/controlling-text-to-image-diffusion-by","title":"Controlling Text-to-Image Diffusion by Orthogonal Finetuning","date":"2023-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zeju1997/oft","path":"oft-control/eval_canny.py","file_url":"https://github.com/zeju1997/oft/blob/HEAD/oft-control/eval_canny.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a73565d61cbda080","mcp_get_code":{"code_sha256":"a73565d61cbda080"}},{"arxiv_id":"2305.14956","paper":"/paper/editing-commonsense-knowledge-in-gpt","title":"Editing Common Sense in Transformers","date":"2023-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anshitag/memit_csk","path":"base_finetune_experiments/fine_tune_gpt.py","file_url":"https://github.com/anshitag/memit_csk/blob/HEAD/base_finetune_experiments/fine_tune_gpt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"41346355360705b1","mcp_get_code":{"code_sha256":"41346355360705b1"}},{"arxiv_id":"2305.14956","paper":"/paper/editing-commonsense-knowledge-in-gpt","title":"Editing Common Sense in Transformers","date":"2023-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anshitag/memit_csk","path":"repair_finetune_experiments/evaluate_affected_finetune_model.py","file_url":"https://github.com/anshitag/memit_csk/blob/HEAD/repair_finetune_experiments/evaluate_affected_finetune_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cfa19d05d0870261","mcp_get_code":{"code_sha256":"cfa19d05d0870261"}},{"arxiv_id":"2212.01588","paper":"/paper/rho-r-reducing-hallucination-in-open-domain","title":"RHO ($ρ$): Reducing Hallucination in Open-domain Dialogues with Knowledge Grounding","date":"2022-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ziweiji/rho","path":"kg-cruse_/codes/KGCruse/predict.py","file_url":"https://github.com/ziweiji/rho/blob/HEAD/kg-cruse_/codes/KGCruse/predict.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dac58534a9991421","mcp_get_code":{"code_sha256":"dac58534a9991421"}},{"arxiv_id":"2211.03846","paper":"/paper/fed-cd-federated-causal-discovery-from","title":"Federated Causal Discovery From Interventions","date":"2022-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aminabyaneh/Federated_CL","path":"federated/utils.py","file_url":"https://github.com/aminabyaneh/Federated_CL/blob/HEAD/federated/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ec867acd8308a4f3","mcp_get_code":{"code_sha256":"ec867acd8308a4f3"}},{"arxiv_id":"2111.05826","paper":"/paper/palette-image-to-image-diffusion-models-1","title":"Palette: Image-to-Image Diffusion Models","date":"2021-11-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kylelo/roofdiffusion","path":"data/util/roof_metric.py","file_url":"https://github.com/kylelo/roofdiffusion/blob/HEAD/data/util/roof_metric.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c9f8bc1be2cb6a09","mcp_get_code":{"code_sha256":"c9f8bc1be2cb6a09"}},{"arxiv_id":"2111.04134","paper":"/paper/mapping-access-to-water-and-sanitation-in","title":"Mapping Access to Water and Sanitation in Colombia using Publicly Accessible Satellite Imagery, Crowd-sourced Geospatial Information and RandomForests","date":null,"month_inferred_from_arxiv_id":"2021-11","title_source":"archive","repo":"thinkingmachines/geoai-immap-wash","path":"utils/modelutils.py","file_url":"https://github.com/thinkingmachines/geoai-immap-wash/blob/HEAD/utils/modelutils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"081ab37dbbcd9611","mcp_get_code":{"code_sha256":"081ab37dbbcd9611"}},{"arxiv_id":"2106.11959","paper":"/paper/revisiting-deep-learning-models-for-tabular","title":"Revisiting Deep Learning Models for Tabular Data","date":"2021-06-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Yura52/tabular-dl-revisiting-models","path":"lib/metrics.py","file_url":"https://github.com/Yura52/tabular-dl-revisiting-models/blob/HEAD/lib/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"12f448bfa8afc6b7","mcp_get_code":{"code_sha256":"12f448bfa8afc6b7"}},{"arxiv_id":"2105.10793","paper":"/paper/goo-a-dataset-for-gaze-object-prediction-in","title":"GOO: A Dataset for Gaze Object Prediction in Retail Environments","date":"2021-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"upeee/GOO-GAZE2021","path":"gazefollowing/evaluate_chong.py","file_url":"https://github.com/upeee/GOO-GAZE2021/blob/HEAD/gazefollowing/evaluate_chong.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5ae5ac5e4843dd4e","mcp_get_code":{"code_sha256":"5ae5ac5e4843dd4e"}},{"arxiv_id":"2105.08445","paper":"/paper/drill-dynamic-representations-for-imbalanced","title":"DRILL: Dynamic Representations for Imbalanced Lifelong Learning","date":"2021-05-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"knowledgetechnologyuhh/drill","path":"models/utils.py","file_url":"https://github.com/knowledgetechnologyuhh/drill/blob/HEAD/models/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"23cde0e185e9c000","mcp_get_code":{"code_sha256":"23cde0e185e9c000"}},{"arxiv_id":"1909.11810","paper":"/paper/mixed-dimension-embeddings-with-application","title":"Mixed Dimension Embeddings with Application to Memory-Efficient Recommendation Systems","date":"2019-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"samiwilf/dlrm_from_shz0116","path":"dlrm_s_caffe2.py","file_url":"https://github.com/samiwilf/dlrm_from_shz0116/blob/HEAD/dlrm_s_caffe2.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"332764c725473589","mcp_get_code":{"code_sha256":"332764c725473589"}},{"arxiv_id":"1909.02107","paper":"/paper/compositional-embeddings-using-complementary","title":"Compositional Embeddings Using Complementary Partitions for Memory-Efficient Recommendation Systems","date":"2019-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"332764c725473589","mcp_get_code":{"code_sha256":"332764c725473589"}},{"arxiv_id":"1906.03109","paper":"/paper/the-architectural-implications-of-facebooks","title":"The Architectural Implications of Facebook's DNN-based Personalized Recommendation","date":"2019-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"332764c725473589","mcp_get_code":{"code_sha256":"332764c725473589"}},{"arxiv_id":"1906.00091","paper":"/paper/190600091","title":"Deep Learning Recommendation Model for Personalization and Recommendation Systems","date":"2019-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"myungkeun-cho/taobao_dlrm","path":"dlrm_s_caffe2.py","file_url":"https://github.com/myungkeun-cho/taobao_dlrm/blob/HEAD/dlrm_s_caffe2.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"332764c725473589","mcp_get_code":{"code_sha256":"332764c725473589"}},{"arxiv_id":"1807.10165","paper":"/paper/unet-a-nested-u-net-architecture-for-medical","title":"UNet++: A Nested U-Net Architecture for Medical Image Segmentation","date":"2018-07-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"marccoru/marinedebrisdetector","path":"marinedebrisdetector/metrics.py","file_url":"https://github.com/marccoru/marinedebrisdetector/blob/HEAD/marinedebrisdetector/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e95088d81cb8fe59","mcp_get_code":{"code_sha256":"e95088d81cb8fe59"}},{"arxiv_id":"1804.04849","paper":"/paper/the-unreasonable-effectiveness-of-the-forget","title":"The unreasonable effectiveness of the forget gate","date":"2018-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JosvanderWesthuizen/janet","path":"aux_code/ops.py","file_url":"https://github.com/JosvanderWesthuizen/janet/blob/HEAD/aux_code/ops.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1f5d644b13f89499","mcp_get_code":{"code_sha256":"1f5d644b13f89499"}},{"arxiv_id":"aaai_35067","paper":null,"title":"arXiv:aaai_35067","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"unicef/giga-global-school-mapping","path":"utils/calib_utils.py","file_url":"https://github.com/unicef/giga-global-school-mapping/blob/HEAD/utils/calib_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3092beb3d90eb468","mcp_get_code":{"code_sha256":"3092beb3d90eb468"}},{"arxiv_id":"2025.findings-emnlp.638","paper":null,"title":"arXiv:2025.findings-emnlp.638","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"liyaooi/LongTableBench","path":"eval/result_process.py","file_url":"https://github.com/liyaooi/LongTableBench/blob/HEAD/eval/result_process.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4f7f74693d581896","mcp_get_code":{"code_sha256":"4f7f74693d581896"}},{"arxiv_id":"2025.acl-long.179","paper":null,"title":"arXiv:2025.acl-long.179","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"RUC-NLPIR/RAG-Critic","path":"rag_error_bench/caculate_acc.py","file_url":"https://github.com/RUC-NLPIR/RAG-Critic/blob/HEAD/rag_error_bench/caculate_acc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd560baa9eec6619","mcp_get_code":{"code_sha256":"bd560baa9eec6619"}}]}