{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/register-dataset","entry":"register_dataset","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":26,"n_papers_ran":8,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":18,"n_samples_ran":5,"n_samples_fingerprinted":0,"n_places":28,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":5,"unverified":13},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.11854","paper":"/paper/arxiv-2606-11854","title":"Fine-tuning Multi-modal LLMs with ART: Art-based Reinforcement Training","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"jinymusim/ART","path":"src/reasoning_with_art/datasets/base.py","file_url":"https://github.com/jinymusim/ART/blob/HEAD/src/reasoning_with_art/datasets/base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"2f1112429b512a86","mcp_get_code":{"code_sha256":"2f1112429b512a86"}},{"arxiv_id":"2605.28420","paper":"/paper/arxiv-2605-28420","title":"Conveyance: A Versatile Framework for Learning in Structured Class Spaces","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"ZKI-PH-ImageAnalysis/Conveyance","path":"MIL/src/conveyance/registry.py","file_url":"https://github.com/ZKI-PH-ImageAnalysis/Conveyance/blob/HEAD/MIL/src/conveyance/registry.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"65fa13105ac5ad94","mcp_get_code":{"code_sha256":"65fa13105ac5ad94"}},{"arxiv_id":"2605.27990","paper":"/paper/arxiv-2605-27990","title":"Geometry-Correct Diffusion Posterior Sampling with Denoiser-Pullback Curvature Guidance and Manifold-Aligned Damping","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Seunghyeok0715/CLAMP","path":"data.py","file_url":"https://github.com/Seunghyeok0715/CLAMP/blob/HEAD/data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"36af44c7119d9fb5","mcp_get_code":{"code_sha256":"36af44c7119d9fb5"}},{"arxiv_id":"2604.20357","paper":"/paper/arxiv-2604-20357","title":"SignDATA: Data Pipeline for Sign Language Translation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"balaboom123/signdata-slt","path":"src/signdata/registry.py","file_url":"https://github.com/balaboom123/signdata-slt/blob/HEAD/src/signdata/registry.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"38b13715850431d1","mcp_get_code":{"code_sha256":"38b13715850431d1"}},{"arxiv_id":"2510.04019","paper":"/paper/arxiv-2510-04019","title":"Simple Policy Gradients for Reasoning with Diffusion Language Models","date":"2025-10-05","month_inferred_from_arxiv_id":null,"title_source":"syntology","repo":"probablyabot/agrpo","path":"data.py","file_url":"https://github.com/probablyabot/agrpo/blob/HEAD/data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e30ee20f00809d83","mcp_get_code":{"code_sha256":"e30ee20f00809d83"}},{"arxiv_id":"2510.03215","paper":"/paper/arxiv-2510-03215","title":"Cache-to-Cache: Direct Semantic Communication Between Large Language Models","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"thu-nics/C2C","path":"rosetta/train/dataset_adapters.py","file_url":"https://github.com/thu-nics/C2C/blob/HEAD/rosetta/train/dataset_adapters.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a335ebc9764afbf4","mcp_get_code":{"code_sha256":"a335ebc9764afbf4"}},{"arxiv_id":"2507.11630","paper":null,"title":"arXiv:2507.11630","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"AlignmentResearch/harmtune","path":"harmtune/datasets.py","file_url":"https://github.com/AlignmentResearch/harmtune/blob/HEAD/harmtune/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c4af8d3c3f7c3a50","mcp_get_code":{"code_sha256":"c4af8d3c3f7c3a50"}},{"arxiv_id":"2505.15962","paper":"/paper/pre-training-large-memory-language-models","title":"Pre-training Large Memory Language Models with Internal and External Knowledge","date":"2025-05-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kilian-group/lmlm","path":"src/lmlm/annotate/dataloader.py","file_url":"https://github.com/kilian-group/lmlm/blob/HEAD/src/lmlm/annotate/dataloader.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a5fa7fed5e977268","mcp_get_code":{"code_sha256":"a5fa7fed5e977268"}},{"arxiv_id":"2502.19960","paper":"/paper/seismollm-advancing-seismic-monitoring-via","title":"SeisMoLLM: Advancing Seismic Monitoring via Cross-modal Transfer with Pre-trained Large Language Model","date":"2025-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"StarMoonWang/SeisMoLLM","path":"datasets/_factory.py","file_url":"https://github.com/StarMoonWang/SeisMoLLM/blob/HEAD/datasets/_factory.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"076ec9da9f9be16a","mcp_get_code":{"code_sha256":"076ec9da9f9be16a"}},{"arxiv_id":"2412.12628","paper":"/paper/dense-audio-visual-event-localization-under","title":"Dense Audio-Visual Event Localization under Cross-Modal Consistency and Multi-Temporal Granularity Collaboration","date":"2024-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zzhhfut/ccnet-aaai2025","path":"libs/datasets/datasets.py","file_url":"https://github.com/zzhhfut/ccnet-aaai2025/blob/HEAD/libs/datasets/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2ee0df0825339a6e","mcp_get_code":{"code_sha256":"2ee0df0825339a6e"}},{"arxiv_id":"2411.18463","paper":"/paper/hotspot-driven-peptide-design-via-multi","title":"Hotspot-Driven Peptide Design via Multi-Fragment Autoregressive Extension","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Ced3-han/PepHAR","path":"datasets/_base.py","file_url":"https://github.com/Ced3-han/PepHAR/blob/HEAD/datasets/_base.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"76294830633c3f09","mcp_get_code":{"code_sha256":"76294830633c3f09"}},{"arxiv_id":"2410.15674","paper":"/paper/talos-enhancing-semantic-scene-completion-via","title":"TALoS: Enhancing Semantic Scene Completion via Test-time Adaptation on the Line of Sight","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"blue-531/TALoS","path":"dataloader/dataset_semantickitti.py","file_url":"https://github.com/blue-531/TALoS/blob/HEAD/dataloader/dataset_semantickitti.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c134fed586d67760","mcp_get_code":{"code_sha256":"c134fed586d67760"}},{"arxiv_id":"2410.15674","paper":"/paper/talos-enhancing-semantic-scene-completion-via","title":"TALoS: Enhancing Semantic Scene Completion via Test-time Adaptation on the Line of Sight","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"blue-531/TALoS","path":"dataloader/pc_dataset.py","file_url":"https://github.com/blue-531/TALoS/blob/HEAD/dataloader/pc_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"06a0e2786c57fc38","mcp_get_code":{"code_sha256":"06a0e2786c57fc38"}},{"arxiv_id":"2407.01521","paper":"/paper/improving-diffusion-inverse-problem-solving","title":"Improving Diffusion Inverse Problem Solving with Decoupled Noise Annealing","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhangbingliang2019/DAPS","path":"data.py","file_url":"https://github.com/zhangbingliang2019/DAPS/blob/HEAD/data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"36af44c7119d9fb5","mcp_get_code":{"code_sha256":"36af44c7119d9fb5"}},{"arxiv_id":"2404.03179","paper":"/paper/uniav-unified-audio-visual-perception-for","title":"UniAV: Unified Audio-Visual Perception for Multi-Task Video Event Localization","date":"2024-04-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ttgeng233/UniAV","path":"libs/datasets/datasets.py","file_url":"https://github.com/ttgeng233/UniAV/blob/HEAD/libs/datasets/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2ee0df0825339a6e","mcp_get_code":{"code_sha256":"2ee0df0825339a6e"}},{"arxiv_id":"2404.02257","paper":"/paper/snag-scalable-and-accurate-video-grounding","title":"SnAG: Scalable and Accurate Video Grounding","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"happyharrycn/actionformer_release","path":"libs/datasets/datasets.py","file_url":"https://github.com/happyharrycn/actionformer_release/blob/HEAD/libs/datasets/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2ee0df0825339a6e","mcp_get_code":{"code_sha256":"2ee0df0825339a6e"}},{"arxiv_id":"2401.14442","paper":"/paper/improving-antibody-humanness-prediction-using","title":"Improving Antibody Humanness Prediction using Patent Data","date":"2024-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AstraZeneca/SelfPAD","path":"utils_finetune/load_data_humanness.py","file_url":"https://github.com/AstraZeneca/SelfPAD/blob/HEAD/utils_finetune/load_data_humanness.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b5e4ef25e7e4854d","mcp_get_code":{"code_sha256":"b5e4ef25e7e4854d"}},{"arxiv_id":"2311.10908","paper":"/paper/equivariant-neural-operator-learning-with-1","title":"Equivariant Neural Operator Learning with Graphon Convolution","date":"2023-11-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ccr-cheng/infgcn-pytorch","path":"datasets/_base.py","file_url":"https://github.com/ccr-cheng/infgcn-pytorch/blob/HEAD/datasets/_base.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"76294830633c3f09","mcp_get_code":{"code_sha256":"76294830633c3f09"}},{"arxiv_id":"2311.03748","paper":"/paper/unified-low-resource-sequence-labeling-by","title":"Unified Low-Resource Sequence Labeling by Sample-Aware Dynamic Sparse Finetuning","date":"2023-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"psunlpgroup/fish-dip","path":"augment/datasets_all.py","file_url":"https://github.com/psunlpgroup/fish-dip/blob/HEAD/augment/datasets_all.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"404087a3ff94d6e0","mcp_get_code":{"code_sha256":"404087a3ff94d6e0"}},{"arxiv_id":"2309.05516","paper":"/paper/optimize-weight-rounding-via-signed-gradient","title":"Optimize Weight Rounding via Signed Gradient Descent for the Quantization of LLMs","date":"2023-09-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"intel/auto-round","path":"auto_round/calib_dataset.py","file_url":"https://github.com/intel/auto-round/blob/HEAD/auto_round/calib_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"34b4d7cc1934723b","mcp_get_code":{"code_sha256":"34b4d7cc1934723b"}},{"arxiv_id":"2305.09731","paper":"/paper/what-in-context-learning-learns-in-context","title":"What In-Context Learning \"Learns\" In-Context: Disentangling Task Recognition and Task Learning","date":"2023-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/whaticllearns","path":"spb/bench_datasets.py","file_url":"https://github.com/princeton-nlp/whaticllearns/blob/HEAD/spb/bench_datasets.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"404087a3ff94d6e0","mcp_get_code":{"code_sha256":"404087a3ff94d6e0"}},{"arxiv_id":"2303.12930","paper":"/paper/dense-localizing-audio-visual-events-in","title":"Dense-Localizing Audio-Visual Events in Untrimmed Videos: A Large-Scale Benchmark and Baseline","date":"2023-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ttgeng233/UnAV","path":"libs/datasets/datasets.py","file_url":"https://github.com/ttgeng233/UnAV/blob/HEAD/libs/datasets/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2ee0df0825339a6e","mcp_get_code":{"code_sha256":"2ee0df0825339a6e"}},{"arxiv_id":"2303.07347","paper":"/paper/tridet-temporal-action-detection-with","title":"TriDet: Temporal Action Detection with Relative Boundary Modeling","date":"2023-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dingfengshi/TriDet","path":"libs/datasets/datasets.py","file_url":"https://github.com/dingfengshi/TriDet/blob/HEAD/libs/datasets/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2ee0df0825339a6e","mcp_get_code":{"code_sha256":"2ee0df0825339a6e"}},{"arxiv_id":"2301.00970","paper":"/paper/benchmarking-the-robustness-of-lidar-semantic","title":"Benchmarking the Robustness of LiDAR Semantic Segmentation Models","date":"2023-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yanx27/2dpass","path":"dataloader/dataset.py","file_url":"https://github.com/yanx27/2dpass/blob/HEAD/dataloader/dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c134fed586d67760","mcp_get_code":{"code_sha256":"c134fed586d67760"}},{"arxiv_id":"2301.00970","paper":"/paper/benchmarking-the-robustness-of-lidar-semantic","title":"Benchmarking the Robustness of LiDAR Semantic Segmentation Models","date":"2023-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yanx27/2dpass","path":"dataloader/pc_dataset.py","file_url":"https://github.com/yanx27/2dpass/blob/HEAD/dataloader/pc_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"06a0e2786c57fc38","mcp_get_code":{"code_sha256":"06a0e2786c57fc38"}},{"arxiv_id":"2205.10475","paper":"/paper/deepstruct-pretraining-of-language-models-for-1","title":"DeepStruct: Pretraining of Language Models for Structure Prediction","date":"2022-05-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cgraywang/deepstruct","path":"src/dataset_processing/datasets.py","file_url":"https://github.com/cgraywang/deepstruct/blob/HEAD/src/dataset_processing/datasets.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"404087a3ff94d6e0","mcp_get_code":{"code_sha256":"404087a3ff94d6e0"}},{"arxiv_id":"openreview_4kdkm56U5b","paper":null,"title":"arXiv:openreview_4kdkm56U5b","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"KurohaneNioko/EADiff","path":"utils/data_processing.py","file_url":"https://github.com/KurohaneNioko/EADiff/blob/HEAD/utils/data_processing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b18982efa98ac3ae","mcp_get_code":{"code_sha256":"b18982efa98ac3ae"}},{"arxiv_id":"Yang_TULIP_Transformer_for_Upsampling_of_LiDAR_Point_Clouds_CVPR_2024_paper","paper":null,"title":"arXiv:Yang_TULIP_Transformer_for_Upsampling_of_LiDAR_Point_Clouds_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ethz-asl/TULIP","path":"tulip/util/datasets.py","file_url":"https://github.com/ethz-asl/TULIP/blob/HEAD/tulip/util/datasets.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43237a496468bfd1","mcp_get_code":{"code_sha256":"43237a496468bfd1"}}]}