{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/data-file-path","entry":"data_file_path","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":9,"n_papers_ran":0,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":1,"n_samples_ran":0,"n_samples_fingerprinted":0,"n_places":9,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":0,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2507.00322","paper":null,"title":"arXiv:2507.00322","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"EleutherAI/pythia","path":"utils/mmap_dataset.py","file_url":"https://github.com/EleutherAI/pythia/blob/HEAD/utils/mmap_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2505.03005","paper":"/paper/radlads-rapid-attention-distillation-to","title":"RADLADS: Rapid Attention Distillation to Linear Attention Decoders at Scale","date":"2025-05-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"recursal/radlads-paper","path":"make_data_hf.py","file_url":"https://github.com/recursal/radlads-paper/blob/HEAD/make_data_hf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2410.17215","paper":"/paper/miniplm-knowledge-distillation-for-pre","title":"MiniPLM: Knowledge Distillation for Pre-Training Language Models","date":"2024-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/MiniPLM","path":"data_utils/distributed_indexed.py","file_url":"https://github.com/thu-coai/MiniPLM/blob/HEAD/data_utils/distributed_indexed.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2402.02625","paper":"/paper/enhancing-transformer-rnns-with-multiple","title":"Enhancing Transformer RNNs with Multiple Temporal Perspectives","date":"2024-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RazvanDu/TemporalRNNs","path":"MultiplePerspectives/src/binidx.py","file_url":"https://github.com/RazvanDu/TemporalRNNs/blob/HEAD/MultiplePerspectives/src/binidx.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2312.02406","paper":"/paper/efficient-online-data-mixing-for-language","title":"Efficient Online Data Mixing For Language Model Pre-Training","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alon-albalak/online-data-mixing","path":"bigram_model.py","file_url":"https://github.com/alon-albalak/online-data-mixing/blob/HEAD/bigram_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2307.15504","paper":"/paper/exploring-format-consistency-for-instruction","title":"Exploring Format Consistency for Instruction Tuning","date":"2023-07-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/unifiedinstructiontuning","path":"model_center/dataset/distributed_indexed.py","file_url":"https://github.com/thunlp/unifiedinstructiontuning/blob/HEAD/model_center/dataset/distributed_indexed.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2305.09137","paper":"/paper/pre-training-to-learn-in-context","title":"Pre-Training to Learn in Context","date":"2023-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/picl","path":"data_utils/distributed_indexed.py","file_url":"https://github.com/thu-coai/picl/blob/HEAD/data_utils/distributed_indexed.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2302.13939","paper":"/paper/spikegpt-generative-pre-trained-language","title":"SpikeGPT: Generative Pre-trained Language Model with Spiking Neural Networks","date":"2023-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ridgerchu/spikegpt","path":"src/binidx.py","file_url":"https://github.com/ridgerchu/spikegpt/blob/HEAD/src/binidx.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}},{"arxiv_id":"2022.emnlp-demos.40","paper":null,"title":"arXiv:2022.emnlp-demos.40","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenBMB/ModelCenter","path":"model_center/dataset/distributed_indexed.py","file_url":"https://github.com/OpenBMB/ModelCenter/blob/HEAD/model_center/dataset/distributed_indexed.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bdf28f9492a770dc","mcp_get_code":{"code_sha256":"bdf28f9492a770dc"}}]}