{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/evaluate-model","entry":"evaluate_model","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":54,"n_papers_ran":18,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":61,"n_samples_ran":21,"n_samples_fingerprinted":1,"n_places":61,"n_places_pointer_only":26,"by_status":{"ran_honours":5,"ran_violates":0,"ran_draft_wrong":4,"ran_fixture":1,"ran":11,"unverified":40},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.18888","paper":"/paper/arxiv-2608-18888","title":"Assessing Quality of Experience in Natural Language Generation of German Text","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"DFKI-NLP/TextQ","path":"models/ats-new_integrating_features.py","file_url":"https://github.com/DFKI-NLP/TextQ/blob/HEAD/models/ats-new_integrating_features.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"130e8363ae0960d0","mcp_get_code":{"code_sha256":"130e8363ae0960d0"}},{"arxiv_id":"2607.06796","paper":"/paper/arxiv-2607-06796","title":"Enhancing deep learning models for time series classification via knowledge distillation","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"MSD-IRIMAS/KD-4-TSC","path":"KD4TSC/utils.py","file_url":"https://github.com/MSD-IRIMAS/KD-4-TSC/blob/HEAD/KD4TSC/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c5bd23970f6c07e2","mcp_get_code":{"code_sha256":"c5bd23970f6c07e2"}},{"arxiv_id":"2607.06638","paper":"/paper/arxiv-2607-06638","title":"UASPL: Uncertainty-Aware Self-Paced Learning with Evidential Neural Networks","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"treelife979/UASPL","path":"uaspl_pic/UASPL_pic.py","file_url":"https://github.com/treelife979/UASPL/blob/HEAD/uaspl_pic/UASPL_pic.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"97ab0b10c1b915e9","mcp_get_code":{"code_sha256":"97ab0b10c1b915e9"}},{"arxiv_id":"2606.14900","paper":"/paper/arxiv-2606-14900","title":"GRASP: Gradient-Aligned Sequential Parameter Transfer for Memory-Efficient Multi-Source Learning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Sekeh-Lab/grasp-multisource-transfer","path":"experiments/grasp/run_grasp_experiment.py","file_url":"https://github.com/Sekeh-Lab/grasp-multisource-transfer/blob/HEAD/experiments/grasp/run_grasp_experiment.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5faf0fe65b442218","mcp_get_code":{"code_sha256":"5faf0fe65b442218"}},{"arxiv_id":"2604.07658","paper":"/paper/arxiv-2604-07658","title":"Optimal Decay Spectra for Linear Recurrences","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"SiLifen/PoST","path":"eval_llm.py","file_url":"https://github.com/SiLifen/PoST/blob/HEAD/eval_llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"16b3e578e4a9214a","mcp_get_code":{"code_sha256":"16b3e578e4a9214a"}},{"arxiv_id":"2603.21970","paper":"/paper/arxiv-2603-21970","title":"Parameter-Efficient Fine-Tuning for Medical Text Summarization: A Comparative Study of Lora, Prompt Tuning, and Full Fine-Tuning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"eracoding/llm-medical-summarization","path":"src/evaluation/evaluator.py","file_url":"https://github.com/eracoding/llm-medical-summarization/blob/HEAD/src/evaluation/evaluator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b569afccc336defc","mcp_get_code":{"code_sha256":"b569afccc336defc"}},{"arxiv_id":"2603.08286","paper":"/paper/arxiv-2603-08286","title":"LAMUS: A Large-Scale Corpus for Legal Argument Mining from U.S. Caselaw using LLMs","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"LavanyaPobbathi/LAMUS","path":"code/experiment/B_finetune_with_legalbench.py","file_url":"https://github.com/LavanyaPobbathi/LAMUS/blob/HEAD/code/experiment/B_finetune_with_legalbench.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5611c734b2eb1c33","mcp_get_code":{"code_sha256":"5611c734b2eb1c33"}},{"arxiv_id":"2602.18406","paper":"/paper/arxiv-2602-18406","title":"GRaM workshop at ICLR 2026 Tiny Paper Track LATENT EQUIVARIANT OPERATORS FOR ROBUST OBJECT RECOGNITION: PROMISES AND CHALLENGES","date":"2026-02-20","month_inferred_from_arxiv_id":null,"title_source":"syntology","repo":"BRAIN-Aalto/equivariant_operator","path":"training/base.py","file_url":"https://github.com/BRAIN-Aalto/equivariant_operator/blob/HEAD/training/base.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a846a651093ddd3a","mcp_get_code":{"code_sha256":"a846a651093ddd3a"}},{"arxiv_id":"2601.18897","paper":"/paper/arxiv-2601-18897","title":"Explainable Uncertainty Quantification for Wastewater Treatment Energy Prediction via Interval Type-2 Neuro-Fuzzy System","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"QusaiKhaled/XUQ","path":"src/experiments/evaluate.py","file_url":"https://github.com/QusaiKhaled/XUQ/blob/HEAD/src/experiments/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a3c2b1b3ce1f57fc","mcp_get_code":{"code_sha256":"a3c2b1b3ce1f57fc"}},{"arxiv_id":"2601.17907","paper":"/paper/arxiv-2601-17907","title":"FARM: Few-shot Adaptive Malware Family Classification under Concept Drift","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"numanhg/farm","path":"src/experiments/build_retrain_dataset.py","file_url":"https://github.com/numanhg/farm/blob/HEAD/src/experiments/build_retrain_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ce0ebe6c7c84c6d8","mcp_get_code":{"code_sha256":"ce0ebe6c7c84c6d8"}},{"arxiv_id":"2601.03429","paper":"/paper/arxiv-2601-03429","title":"DeepLeak: Privacy Enhancing Hardening of Model Explanations Against Membership Leakage","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"um-dsp/DeepLeak","path":"utils/evaluation.py","file_url":"https://github.com/um-dsp/DeepLeak/blob/HEAD/utils/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9050ce24d753dc98","mcp_get_code":{"code_sha256":"9050ce24d753dc98"}},{"arxiv_id":"2511.19635","paper":"/paper/arxiv-2511-19635","title":"Agint: Agentic Graph Compilation for Software Engineering Agents","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"AgintHub/nifty-wilson","path":"outputs/dagify/train_large_language_model/code/evaluate_model.py","file_url":"https://github.com/AgintHub/nifty-wilson/blob/HEAD/outputs/dagify/train_large_language_model/code/evaluate_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4f3f15762881b44c","mcp_get_code":{"code_sha256":"4f3f15762881b44c"}},{"arxiv_id":"2511.03824","paper":"/paper/arxiv-2511-03824","title":"Sketch-Augmented Features Improve Learning Long-Range Dependencies in Graph Neural Networks","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"ryienh/sketched-random-features","path":"srf/oversmoothing.py","file_url":"https://github.com/ryienh/sketched-random-features/blob/HEAD/srf/oversmoothing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e430efc7cb3c7a13","mcp_get_code":{"code_sha256":"e430efc7cb3c7a13"}},{"arxiv_id":"2509.24372","paper":"/paper/arxiv-2509-24372","title":"Evolution Strategies at Scale: LLM Fine-Tuning Beyond Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"VsonicV/es-at-scale","path":"archive/es_fine-tuning_conciseness.py","file_url":"https://github.com/VsonicV/es-at-scale/blob/HEAD/archive/es_fine-tuning_conciseness.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"f50eda6a0c64b38b","mcp_get_code":{"code_sha256":"f50eda6a0c64b38b"}},{"arxiv_id":"2509.15786","paper":"/paper/arxiv-2509-15786","title":"Building Data-Driven Occupation Taxonomies: A Bottom-Up Multi-Stage Approach via Semantic Clustering and Multi-Agent Collaboration","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"aida-ugent/CLIMB","path":"src/train_classifier.py","file_url":"https://github.com/aida-ugent/CLIMB/blob/HEAD/src/train_classifier.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"af8114aa86e2379e","mcp_get_code":{"code_sha256":"af8114aa86e2379e"}},{"arxiv_id":"2509.14030","paper":"/paper/arxiv-2509-14030","title":"CrowdAgent: Multi-Agent Managed Multi-Source Annotation System","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"QMMMS/CrowdAgent","path":"crowd-agent-backend/src/slm_function/robust_convnextv2.py","file_url":"https://github.com/QMMMS/CrowdAgent/blob/HEAD/crowd-agent-backend/src/slm_function/robust_convnextv2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2761c61217ad51cb","mcp_get_code":{"code_sha256":"2761c61217ad51cb"}},{"arxiv_id":"2506.16950","paper":null,"title":"arXiv:2506.16950","date":null,"month_inferred_from_arxiv_id":"2025-06","title_source":null,"repo":"FanfeiLi/LAION-C","path":"laionc/evaluation.py","file_url":"https://github.com/FanfeiLi/LAION-C/blob/HEAD/laionc/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"28f422bd4478f183","mcp_get_code":{"code_sha256":"28f422bd4478f183"}},{"arxiv_id":"2504.02199","paper":"/paper/esc-erasing-space-concept-for-knowledge","title":"ESC: Erasing Space Concept for Knowledge Deletion","date":"2025-04-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KU-VGI/ESC","path":"unlearn.py","file_url":"https://github.com/KU-VGI/ESC/blob/HEAD/unlearn.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8083bb6afa3f1008","mcp_get_code":{"code_sha256":"8083bb6afa3f1008"}},{"arxiv_id":"2502.00270","paper":"/paper/duet-optimizing-training-data-mixtures-via","title":"DUET: Optimizing Training Data Mixtures via Feedback from Unseen Evaluation Tasks","date":"2025-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pmsdapfmbf/duet","path":"BO.py","file_url":"https://github.com/pmsdapfmbf/duet/blob/HEAD/BO.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2642938bca13a388","mcp_get_code":{"code_sha256":"2642938bca13a388"}},{"arxiv_id":"2412.18319","paper":"/paper/mulberry-empowering-mllm-with-o1-like","title":"Mulberry: Empowering MLLM with o1-like Reasoning and Reflection via Collective Monte Carlo Tree Search","date":"2024-12-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"phunterlau/paper_without_code","path":"output/workflow/2411.16905/generated_code.py","file_url":"https://github.com/phunterlau/paper_without_code/blob/HEAD/output/workflow/2411.16905/generated_code.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b20fc8d9642b452c","mcp_get_code":{"code_sha256":"b20fc8d9642b452c"}},{"arxiv_id":"2411.00056","paper":null,"title":"arXiv:2411.00056","date":null,"month_inferred_from_arxiv_id":"2024-11","title_source":null,"repo":"DarianRodriguez/NegVerse","path":"negator/Evaluation/evaluate.py","file_url":"https://github.com/DarianRodriguez/NegVerse/blob/HEAD/negator/Evaluation/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5fd9b03e5e82052a","mcp_get_code":{"code_sha256":"5fd9b03e5e82052a"}},{"arxiv_id":"2410.11289","paper":"/paper/subspace-optimization-for-large-language","title":"Subspace Optimization for Large Language Models with Convergence Guarantees","date":"2024-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pkumelon/Golore","path":"torchrun_main.py","file_url":"https://github.com/pkumelon/Golore/blob/HEAD/torchrun_main.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"f33e66905cc8abcb","mcp_get_code":{"code_sha256":"f33e66905cc8abcb"}},{"arxiv_id":"2410.10318","paper":"/paper/qianets-quantum-integrated-adaptive-networks","title":"QIANets: Quantum-Integrated Adaptive Networks for Reduced Latency and Improved Inference Times in CNN Models","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"edwardmagongo/quantum-inspired-model-compression","path":"densenet.py","file_url":"https://github.com/edwardmagongo/quantum-inspired-model-compression/blob/HEAD/densenet.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"66d2c815d123a991","mcp_get_code":{"code_sha256":"66d2c815d123a991"}},{"arxiv_id":"2410.10254","paper":"/paper/lolcats-on-low-rank-linearizing-of-large","title":"LoLCATs: On Low-Rank Linearizing of Large Language Models","date":"2024-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HazyResearch/lolcats","path":"distill_llama.py","file_url":"https://github.com/HazyResearch/lolcats/blob/HEAD/distill_llama.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"99199bb3bba7f5ed","mcp_get_code":{"code_sha256":"99199bb3bba7f5ed"}},{"arxiv_id":"2410.06296","paper":"/paper/conformal-structured-prediction","title":"Conformal Structured Prediction","date":"2024-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"botong516/conformal_structured_prediction","path":"run/imagenet.py","file_url":"https://github.com/botong516/conformal_structured_prediction/blob/HEAD/run/imagenet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"edbfb59fb3879819","mcp_get_code":{"code_sha256":"edbfb59fb3879819"}},{"arxiv_id":"2410.02958","paper":"/paper/automl-agent-a-multi-agent-llm-framework-for","title":"AutoML-Agent: A Multi-Agent LLM Framework for Full-Pipeline AutoML","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DeepAuto-AI/automl-agent","path":"prompt_pool/image_classification.py","file_url":"https://github.com/DeepAuto-AI/automl-agent/blob/HEAD/prompt_pool/image_classification.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"85044f46d9712e2b","mcp_get_code":{"code_sha256":"85044f46d9712e2b"}},{"arxiv_id":"2410.02958","paper":"/paper/automl-agent-a-multi-agent-llm-framework-for","title":"AutoML-Agent: A Multi-Agent LLM Framework for Full-Pipeline AutoML","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DeepAuto-AI/automl-agent","path":"prompt_pool/node_classification.py","file_url":"https://github.com/DeepAuto-AI/automl-agent/blob/HEAD/prompt_pool/node_classification.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e923c7c2b3aea681","mcp_get_code":{"code_sha256":"e923c7c2b3aea681"}},{"arxiv_id":"2409.19663","paper":"/paper/identifying-knowledge-editing-types-in-large","title":"Identifying Knowledge Editing Types in Large Language Models","date":"2024-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xpq-tech/keti","path":"abalation_closellm.py","file_url":"https://github.com/xpq-tech/keti/blob/HEAD/abalation_closellm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"91a84b1a09f3183e","mcp_get_code":{"code_sha256":"91a84b1a09f3183e"}},{"arxiv_id":"2409.19663","paper":"/paper/identifying-knowledge-editing-types-in-large","title":"Identifying Knowledge Editing Types in Large Language Models","date":"2024-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xpq-tech/keti","path":"close_source_llms_main.py","file_url":"https://github.com/xpq-tech/keti/blob/HEAD/close_source_llms_main.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"611b5daad6c596b7","mcp_get_code":{"code_sha256":"611b5daad6c596b7"}},{"arxiv_id":"2409.04318","paper":"/paper/learning-vs-retrieval-the-role-of-in-context","title":"Learning vs Retrieval: The Role of In-Context Examples in Regression with LLMs","date":"2024-09-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HLR/LvsR-LLM","path":"Datasets/model_training.py","file_url":"https://github.com/HLR/LvsR-LLM/blob/HEAD/Datasets/model_training.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ee73f674640a4661","mcp_get_code":{"code_sha256":"ee73f674640a4661"}},{"arxiv_id":"2405.04517","paper":"/paper/xlstm-extended-long-short-term-memory","title":"xLSTM: Extended Long Short-Term Memory","date":"2024-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gonzalopezgil/xlstm-ts","path":"src/ml/models/xlstm_ts/training.py","file_url":"https://github.com/gonzalopezgil/xlstm-ts/blob/HEAD/src/ml/models/xlstm_ts/training.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"eefa6a8362a8958c","mcp_get_code":{"code_sha256":"eefa6a8362a8958c"}},{"arxiv_id":"2403.07721","paper":"/paper/visual-decoding-and-reconstruction-via-eeg","title":"Visual Decoding and Reconstruction via EEG Embeddings with Guided Diffusion","date":null,"month_inferred_from_arxiv_id":"2024-03","title_source":"archive","repo":"dongyangli-del/eeg_image_decode","path":"Generation/ATMS_reconstruction.py","file_url":"https://github.com/dongyangli-del/eeg_image_decode/blob/HEAD/Generation/ATMS_reconstruction.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dde08c57d862ec73","mcp_get_code":{"code_sha256":"dde08c57d862ec73"}},{"arxiv_id":"2403.07721","paper":"/paper/visual-decoding-and-reconstruction-via-eeg","title":"Visual Decoding and Reconstruction via EEG Embeddings with Guided Diffusion","date":null,"month_inferred_from_arxiv_id":"2024-03","title_source":"archive","repo":"dongyangli-del/eeg_image_decode","path":"Retrieval/ATMS_retrieval.py","file_url":"https://github.com/dongyangli-del/eeg_image_decode/blob/HEAD/Retrieval/ATMS_retrieval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a8f9f460931bbeb6","mcp_get_code":{"code_sha256":"a8f9f460931bbeb6"}},{"arxiv_id":"2403.07721","paper":"/paper/visual-decoding-and-reconstruction-via-eeg","title":"Visual Decoding and Reconstruction via EEG Embeddings with Guided Diffusion","date":null,"month_inferred_from_arxiv_id":"2024-03","title_source":"archive","repo":"nzwang/neural-mcrl","path":"EEGToVisual/NeuralMCRL.py","file_url":"https://github.com/nzwang/neural-mcrl/blob/HEAD/EEGToVisual/NeuralMCRL.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1979eed793bbbbf9","mcp_get_code":{"code_sha256":"1979eed793bbbbf9"}},{"arxiv_id":"2403.00963","paper":"/paper/tree-regularized-tabular-embeddings","title":"Tree-Regularized Tabular Embeddings","date":"2024-03-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"milanlx/tree-regularized-embedding","path":"src/train_tree.py","file_url":"https://github.com/milanlx/tree-regularized-embedding/blob/HEAD/src/train_tree.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ecfcea6bb0afa1b9","mcp_get_code":{"code_sha256":"ecfcea6bb0afa1b9"}},{"arxiv_id":"2402.02823","paper":"/paper/evading-data-contamination-detection-for","title":"Evading Data Contamination Detection for Language Models is (too) Easy","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eth-sri/malicious-contamination","path":"code-contamination-detection/src/utils.py","file_url":"https://github.com/eth-sri/malicious-contamination/blob/HEAD/code-contamination-detection/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"01d6eba772ddf7c2","mcp_get_code":{"code_sha256":"01d6eba772ddf7c2"}},{"arxiv_id":"2401.17230","paper":"/paper/espnet-spk-full-pipeline-speaker-embedding","title":"ESPnet-SPK: full pipeline speaker embedding toolkit with reproducible recipes, self-supervised front-ends, and off-the-shelf models","date":"2024-01-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Jungjee/RawNet","path":"python/RawNet1/PyTorch/train_RawNet.py","file_url":"https://github.com/Jungjee/RawNet/blob/HEAD/python/RawNet1/PyTorch/train_RawNet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bb8e18446e6e5c44","mcp_get_code":{"code_sha256":"bb8e18446e6e5c44"}},{"arxiv_id":"2401.05952","paper":"/paper/llm-as-a-coauthor-the-challenges-of-detecting","title":"LLM-as-a-Coauthor: Can Mixed Human-Written and Machine-Generated Text Be Detected?","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongping-chen/mixset","path":"methods/radar.py","file_url":"https://github.com/dongping-chen/mixset/blob/HEAD/methods/radar.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"23449ffb5d7ab7b0","mcp_get_code":{"code_sha256":"23449ffb5d7ab7b0"}},{"arxiv_id":"2401.04148","paper":"/paper/online-test-time-adaptation-of-spatial","title":"Online Test-Time Adaptation of Spatial-Temporal Traffic Flow Forecasting","date":"2024-01-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pengxin-guo/adcsd","path":"libcity/evaluator/utils.py","file_url":"https://github.com/pengxin-guo/adcsd/blob/HEAD/libcity/evaluator/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9b199ea651c27032","mcp_get_code":{"code_sha256":"9b199ea651c27032"}},{"arxiv_id":"2312.12112","paper":"/paper/curated-llm-synergy-of-llms-and-data-curation","title":"Curated LLM: Synergy of LLMs and Data Curation for tabular augmentation in low-data regimes","date":"2023-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vanderschaarlab/cllm","path":"src/cllm/utils.py","file_url":"https://github.com/vanderschaarlab/cllm/blob/HEAD/src/cllm/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8097d26afff455d8","mcp_get_code":{"code_sha256":"8097d26afff455d8"}},{"arxiv_id":"2310.08446","paper":"/paper/towards-robust-multi-modal-reasoning-via","title":"Towards Robust Multi-Modal Reasoning via Model Selection","date":"2023-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LINs-lab/M3","path":"MS-GQA/code/run_m3.py","file_url":"https://github.com/LINs-lab/M3/blob/HEAD/MS-GQA/code/run_m3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a1cb3bb3f6e48673","mcp_get_code":{"code_sha256":"a1cb3bb3f6e48673"}},{"arxiv_id":"2310.08446","paper":"/paper/towards-robust-multi-modal-reasoning-via","title":"Towards Robust Multi-Modal Reasoning via Model Selection","date":"2023-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LINs-lab/M3","path":"MS-GQA/code/run_ncf++.py","file_url":"https://github.com/LINs-lab/M3/blob/HEAD/MS-GQA/code/run_ncf%2B%2B.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"80cf4f2a02aa0335","mcp_get_code":{"code_sha256":"80cf4f2a02aa0335"}},{"arxiv_id":"2308.12532","paper":"/paper/fedsol-bridging-global-alignment-and-local","title":"FedSOL: Stabilized Orthogonal Learning with Proximal Restrictions in Federated Learning","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Lee-Gihun/FedSOL","path":"algorithms/measures.py","file_url":"https://github.com/Lee-Gihun/FedSOL/blob/HEAD/algorithms/measures.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"aa1b8a2c60165efc","mcp_get_code":{"code_sha256":"aa1b8a2c60165efc"}},{"arxiv_id":"2307.05695","paper":"/paper/stack-more-layers-differently-high-rank","title":"ReLoRA: High-Rank Training Through Low-Rank Updates","date":"2023-07-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guitaricet/relora","path":"torchrun_main.py","file_url":"https://github.com/guitaricet/relora/blob/HEAD/torchrun_main.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6cd1023ec2204c22","mcp_get_code":{"code_sha256":"6cd1023ec2204c22"}},{"arxiv_id":"2306.05093","paper":"/paper/re-aligning-shadow-models-can-improve-white","title":"Investigating the Effect of Misalignment on Membership Privacy in the White-box Setting","date":"2023-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/shadow-realignment-mia","path":"train_controlled_randomness.py","file_url":"https://github.com/microsoft/shadow-realignment-mia/blob/HEAD/train_controlled_randomness.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3528a0b4188e1ed8","mcp_get_code":{"code_sha256":"3528a0b4188e1ed8"}},{"arxiv_id":"2306.03819","paper":"/paper/leace-perfect-linear-concept-erasure-in","title":"LEACE: Perfect linear concept erasure in closed form","date":"2023-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FairUnlearn/detoxai","path":"src/detoxai/core/evaluation.py","file_url":"https://github.com/FairUnlearn/detoxai/blob/HEAD/src/detoxai/core/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"375ff94b7f4a2397","mcp_get_code":{"code_sha256":"375ff94b7f4a2397"}},{"arxiv_id":"2210.13043","paper":"/paper/data-iq-characterizing-subgroups-with","title":"Data-IQ: Characterizing subgroups with heterogeneous outcomes in tabular data","date":"2022-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seedatnabeel/data-iq","path":"src/utils/group_dro_helpers.py","file_url":"https://github.com/seedatnabeel/data-iq/blob/HEAD/src/utils/group_dro_helpers.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"74b9aada2ba5b3c6","mcp_get_code":{"code_sha256":"74b9aada2ba5b3c6"}},{"arxiv_id":"2207.06343","paper":"/paper/tct-convexifying-federated-learning-using","title":"TCT: Convexifying Federated Learning using Bootstrapped Neural Tangent Kernels","date":"2022-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yaodongyu/tct","path":"mnist/utils.py","file_url":"https://github.com/yaodongyu/tct/blob/HEAD/mnist/utils.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bca65ccbc3175672","mcp_get_code":{"code_sha256":"bca65ccbc3175672"}},{"arxiv_id":"2207.02842","paper":"/paper/when-does-bias-transfer-in-transfer-learning","title":"When does Bias Transfer in Transfer Learning?","date":"2022-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MadryLab/bias-transfer","path":"src/eval_utils.py","file_url":"https://github.com/MadryLab/bias-transfer/blob/HEAD/src/eval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8e3e794b7f3dd5f7","mcp_get_code":{"code_sha256":"8e3e794b7f3dd5f7"}},{"arxiv_id":"2206.02909","paper":"/paper/self-supervised-learning-for-human-activity","title":"Self-supervised Learning for Human Activity Recognition Using 700,000 Person-days of Wearable Data","date":"2022-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OxWearables/ssl-wearables","path":"mtl.py","file_url":"https://github.com/OxWearables/ssl-wearables/blob/HEAD/mtl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"6a84d2950d1a7b4c","mcp_get_code":{"code_sha256":"6a84d2950d1a7b4c"}},{"arxiv_id":"2206.02909","paper":"/paper/self-supervised-learning-for-human-activity","title":"Self-supervised Learning for Human Activity Recognition Using 700,000 Person-days of Wearable Data","date":"2022-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OxWearables/ssl-wearables","path":"downstream_task_evaluation.py","file_url":"https://github.com/OxWearables/ssl-wearables/blob/HEAD/downstream_task_evaluation.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a8d567ee48f89c59","mcp_get_code":{"code_sha256":"a8d567ee48f89c59"}},{"arxiv_id":"2204.12632","paper":"/paper/testing-the-ability-of-language-models-to-1","title":"Testing the Ability of Language Models to Interpret Figurative Language","date":"2022-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nightingal3/fig-qa","path":"src/models/gpt_score.py","file_url":"https://github.com/nightingal3/fig-qa/blob/HEAD/src/models/gpt_score.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2a5615f598f3af2b","mcp_get_code":{"code_sha256":"2a5615f598f3af2b"}},{"arxiv_id":"2110.07607","paper":"/paper/humbugdb-a-large-scale-acoustic-mosquito","title":"HumBugDB: A Large-scale Acoustic Mosquito Dataset","date":"2021-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HumBug-Mosquito/HumBugDB","path":"lib/Keras/runKeras.py","file_url":"https://github.com/HumBug-Mosquito/HumBugDB/blob/HEAD/lib/Keras/runKeras.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fffbb3c08c0354be","mcp_get_code":{"code_sha256":"fffbb3c08c0354be"}},{"arxiv_id":"2107.08221","paper":"/paper/visual-representation-learning-does-not","title":"Visual Representation Learning Does Not Generalize Strongly Within the Same Domain","date":"2021-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bethgelab/InDomainGeneralizationBenchmark","path":"src/lablet_generalization_benchmark/evaluate_model.py","file_url":"https://github.com/bethgelab/InDomainGeneralizationBenchmark/blob/HEAD/src/lablet_generalization_benchmark/evaluate_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"84a0f2cd3abb6e6f","mcp_get_code":{"code_sha256":"84a0f2cd3abb6e6f"}},{"arxiv_id":"2006.08852","paper":"/paper/counterexample-guided-learning-of-monotonic","title":"Counterexample-Guided Learning of Monotonic Neural Networks","date":"2020-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AishwaryaSivaraman/COMET","path":"src/ModelCalls.py","file_url":"https://github.com/AishwaryaSivaraman/COMET/blob/HEAD/src/ModelCalls.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"357f326cc7f53e40","mcp_get_code":{"code_sha256":"357f326cc7f53e40"}},{"arxiv_id":"2006.04727","paper":"/paper/theoretical-guarantees-for-learning","title":"Neural Jump Ordinary Differential Equations: Consistent Continuous-Time Prediction and Filtering","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HerreraKrachTeichmann/ControlledODERNN","path":"NJODE/physionet_train.py","file_url":"https://github.com/HerreraKrachTeichmann/ControlledODERNN/blob/HEAD/NJODE/physionet_train.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e7c9692657b50119","mcp_get_code":{"code_sha256":"e7c9692657b50119"}},{"arxiv_id":"2006.04727","paper":"/paper/theoretical-guarantees-for-learning","title":"Neural Jump Ordinary Differential Equations: Consistent Continuous-Time Prediction and Filtering","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HerreraKrachTeichmann/ControlledODERNN","path":"NJODE/climate_train.py","file_url":"https://github.com/HerreraKrachTeichmann/ControlledODERNN/blob/HEAD/NJODE/climate_train.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"172c6c49b142f584","mcp_get_code":{"code_sha256":"172c6c49b142f584"}},{"arxiv_id":"2003.02228","paper":"/paper/pushnet-efficient-and-adaptive-neural-message","title":"PushNet: Efficient and Adaptive Neural Message Passing","date":"2020-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"buschju/pushnet","path":"src/training.py","file_url":"https://github.com/buschju/pushnet/blob/HEAD/src/training.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"03b3243eaff8a95c","mcp_get_code":{"code_sha256":"03b3243eaff8a95c"}},{"arxiv_id":"1910.01179","paper":"/paper/learning-calibratable-policies-using","title":"Learning Calibratable Policies using Programmatic Style-Consistency","date":"2019-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kushaltirumala/callibratable_style_consistency_flies","path":"eval_model.py","file_url":"https://github.com/kushaltirumala/callibratable_style_consistency_flies/blob/HEAD/eval_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f186f46f003e4331","mcp_get_code":{"code_sha256":"f186f46f003e4331"}},{"arxiv_id":"1901.07441","paper":"/paper/padchest-a-large-chest-x-ray-image-dataset","title":"PadChest: A large chest x-ray image dataset with multi-label annotated reports","date":"2019-01-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rajpurkarlab/chexzero","path":"metrics.py","file_url":"https://github.com/rajpurkarlab/chexzero/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5dba521f400b96c2","mcp_get_code":{"code_sha256":"5dba521f400b96c2"}},{"arxiv_id":"2025.emnlp-main.472","paper":null,"title":"arXiv:2025.emnlp-main.472","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"FKarl/HYDRA","path":"hydra_experiments.py","file_url":"https://github.com/FKarl/HYDRA/blob/HEAD/hydra_experiments.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"67890948491fe2e0","mcp_get_code":{"code_sha256":"67890948491fe2e0"}}]}