{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extend-and-repeat","entry":"extend_and_repeat","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":10,"n_papers_ran":10,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":3,"n_samples_ran":3,"n_samples_fingerprinted":2,"n_places":11,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":1,"unverified":0},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2412.18855","paper":"/paper/optimistic-critic-reconstruction-and","title":"Optimistic Critic Reconstruction and Constrained Fine-Tuning for General Offline-to-Online RL","date":"2024-12-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QinwenLuo/OCR-CFT","path":"net.py","file_url":"https://github.com/QinwenLuo/OCR-CFT/blob/HEAD/net.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a4d39c2adb56e8ce","mcp_get_code":{"code_sha256":"a4d39c2adb56e8ce"}},{"arxiv_id":"2405.16907","paper":"/paper/gta-generative-trajectory-augmentation-with","title":"GTA: Generative Trajectory Augmentation with Guidance for Offline Reinforcement Learning","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Jaewoopudding/GTA","path":"corl/algorithms/cql.py","file_url":"https://github.com/Jaewoopudding/GTA/blob/HEAD/corl/algorithms/cql.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4d39c2adb56e8ce","mcp_get_code":{"code_sha256":"a4d39c2adb56e8ce"}},{"arxiv_id":"2405.05349","paper":"/paper/offline-model-based-optimization-via-policy","title":"Offline Model-Based Optimization via Policy-Guided Gradient Search","date":"2024-05-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yassineCh/PGS","path":"model.py","file_url":"https://github.com/yassineCh/PGS/blob/HEAD/model.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1e5f5f635bdf773a","mcp_get_code":{"code_sha256":"1e5f5f635bdf773a"}},{"arxiv_id":"2310.08558","paper":"/paper/offline-retraining-for-online-rl-decoupled","title":"Offline Retraining for Online RL: Decoupled Policy Learning to Mitigate Exploration Bias","date":"2023-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MaxSobolMark/OOO","path":"OOO_for_calql/jax_utils.py","file_url":"https://github.com/MaxSobolMark/OOO/blob/HEAD/OOO_for_calql/jax_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d2f199c948f2a9ed","mcp_get_code":{"code_sha256":"d2f199c948f2a9ed"}},{"arxiv_id":"2309.10790","paper":null,"title":"arXiv:2309.10790","date":null,"month_inferred_from_arxiv_id":"2023-09","title_source":null,"repo":"csmile-1006/ARP","path":"arp_dt/utils.py","file_url":"https://github.com/csmile-1006/ARP/blob/HEAD/arp_dt/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2f199c948f2a9ed","mcp_get_code":{"code_sha256":"d2f199c948f2a9ed"}},{"arxiv_id":"2210.13431","paper":"/paper/instruction-following-agents-with-jointly-pre","title":"Instruction-Following Agents with Multimodal Transformer","date":"2022-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lhao499/instructrl","path":"instructrl/utils.py","file_url":"https://github.com/lhao499/instructrl/blob/HEAD/instructrl/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2f199c948f2a9ed","mcp_get_code":{"code_sha256":"d2f199c948f2a9ed"}},{"arxiv_id":"2210.07484","paper":"/paper/mutual-information-regularized-offline-1","title":"Mutual Information Regularized Offline Reinforcement Learning","date":"2022-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sail-sg/MISA","path":"utilities/jax_utils.py","file_url":"https://github.com/sail-sg/MISA/blob/HEAD/utilities/jax_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2f199c948f2a9ed","mcp_get_code":{"code_sha256":"d2f199c948f2a9ed"}},{"arxiv_id":"2210.06518","paper":"/paper/semi-supervised-offline-reinforcement-1","title":"Semi-Supervised Offline Reinforcement Learning with Action-Free Trajectories","date":"2022-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scottemmons/youngs-cql","path":"SimpleSAC/model.py","file_url":"https://github.com/scottemmons/youngs-cql/blob/HEAD/SimpleSAC/model.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1e5f5f635bdf773a","mcp_get_code":{"code_sha256":"1e5f5f635bdf773a"}},{"arxiv_id":"2207.10050","paper":"/paper/discriminator-weighted-offline-imitation-1","title":"Discriminator-Weighted Offline Imitation Learning from Suboptimal Demonstrations","date":"2022-07-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zexusun/oilca-neurips23","path":"algos/augment_DWBC.py","file_url":"https://github.com/zexusun/oilca-neurips23/blob/HEAD/algos/augment_DWBC.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1e5f5f635bdf773a","mcp_get_code":{"code_sha256":"1e5f5f635bdf773a"}},{"arxiv_id":"2006.04779","paper":"/paper/conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"young-geng/cql","path":"SimpleSAC/model.py","file_url":"https://github.com/young-geng/cql/blob/HEAD/SimpleSAC/model.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1e5f5f635bdf773a","mcp_get_code":{"code_sha256":"1e5f5f635bdf773a"}},{"arxiv_id":"2006.04779","paper":"/paper/conservative-q-learning-for-offline","title":"Conservative Q-Learning for Offline Reinforcement Learning","date":"2020-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"young-geng/jaxcql","path":"JaxCQL/jax_utils.py","file_url":"https://github.com/young-geng/jaxcql/blob/HEAD/JaxCQL/jax_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d2f199c948f2a9ed","mcp_get_code":{"code_sha256":"d2f199c948f2a9ed"}}]}