{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/target-update","entry":"target_update","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":9,"n_papers_ran":1,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":10,"n_samples_ran":1,"n_samples_fingerprinted":0,"n_places":10,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":0,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.11711","paper":"/paper/arxiv-2605-11711","title":"Debiased Model-based Representations for Sample-efficient Continuous Control","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"nothingbutbut/FoG","path":"jaxrl/agents/obac/obac_learner.py","file_url":"https://github.com/nothingbutbut/FoG/blob/HEAD/jaxrl/agents/obac/obac_learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a183b9c35691ddd9","mcp_get_code":{"code_sha256":"a183b9c35691ddd9"}},{"arxiv_id":"2406.00645","paper":"/paper/furl-visual-language-models-as-fuzzy-rewards","title":"FuRL: Visual-Language Models as Fuzzy Rewards for Reinforcement Learning","date":"2024-06-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fuyw/furl","path":"models/furl.py","file_url":"https://github.com/fuyw/furl/blob/HEAD/models/furl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"88369aa7aff57906","mcp_get_code":{"code_sha256":"88369aa7aff57906"}},{"arxiv_id":"2405.16158","paper":"/paper/bigger-regularized-optimistic-scaling-for","title":"Bigger, Regularized, Optimistic: scaling for compute and sample-efficient continuous control","date":"2024-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"proceduralia/high_replay_ratio_continuous_control","path":"jaxrl/agents/sac/sac_learner.py","file_url":"https://github.com/proceduralia/high_replay_ratio_continuous_control/blob/HEAD/jaxrl/agents/sac/sac_learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"804f231f90e60d49","mcp_get_code":{"code_sha256":"804f231f90e60d49"}},{"arxiv_id":"2405.16158","paper":"/paper/bigger-regularized-optimistic-scaling-for","title":"Bigger, Regularized, Optimistic: scaling for compute and sample-efficient continuous control","date":"2024-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naumix/BiggerRegularizedOptimistic","path":"jaxrl/bro/bro_learner.py","file_url":"https://github.com/naumix/BiggerRegularizedOptimistic/blob/HEAD/jaxrl/bro/bro_learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ce51f9f65a9e0a89","mcp_get_code":{"code_sha256":"ce51f9f65a9e0a89"}},{"arxiv_id":"2405.04342","paper":"/paper/the-curse-of-diversity-in-ensemble-based","title":"The Curse of Diversity in Ensemble-Based Exploration","date":"2024-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhixuan-lin/ensemble-rl-continuous","path":"jaxrl/agents/cel/cel_learner.py","file_url":"https://github.com/zhixuan-lin/ensemble-rl-continuous/blob/HEAD/jaxrl/agents/cel/cel_learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d7b5436efe6a474c","mcp_get_code":{"code_sha256":"d7b5436efe6a474c"}},{"arxiv_id":"2402.17135","paper":"/paper/unsupervised-zero-shot-reinforcement-learning","title":"Unsupervised Zero-Shot Reinforcement Learning via Functional Reward Encodings","date":"2024-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kvfrans/fre","path":"common/train_state.py","file_url":"https://github.com/kvfrans/fre/blob/HEAD/common/train_state.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b10ceb550bad20ce","mcp_get_code":{"code_sha256":"b10ceb550bad20ce"}},{"arxiv_id":"2310.17966","paper":"/paper/train-once-get-a-family-state-adaptive-1","title":"Train Once, Get a Family: State-Adaptive Balances for Offline-to-Online Reinforcement Learning","date":"2023-10-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leaplabthu/famo2o","path":"jax_iql/family_learner.py","file_url":"https://github.com/leaplabthu/famo2o/blob/HEAD/jax_iql/family_learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"958fa187e9bfa7b1","mcp_get_code":{"code_sha256":"958fa187e9bfa7b1"}},{"arxiv_id":"2210.04251","paper":"/paper/state-advantage-weighting-for-offline-rl","title":"State Advantage Weighting for Offline RL","date":"2022-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ikostrikov/implicit_q_learning","path":"learner.py","file_url":"https://github.com/ikostrikov/implicit_q_learning/blob/HEAD/learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"87e5a6628edb14db","mcp_get_code":{"code_sha256":"87e5a6628edb14db"}},{"arxiv_id":"2110.01528","paper":"/paper/large-batch-experience-replay","title":"Large Batch Experience Replay","date":"2021-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xavierchanglingli/regularized-optimal-experience-replay","path":"sac_learner.py","file_url":"https://github.com/xavierchanglingli/regularized-optimal-experience-replay/blob/HEAD/sac_learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e85be23fbac3d092","mcp_get_code":{"code_sha256":"e85be23fbac3d092"}},{"arxiv_id":"1801.01290","paper":"/paper/soft-actor-critic-off-policy-maximum-entropy","title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","date":"2018-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ikostrikov/jax-rl","path":"jaxrl/agents/sac/sac_learner.py","file_url":"https://github.com/ikostrikov/jax-rl/blob/HEAD/jaxrl/agents/sac/sac_learner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f62b5d3c6983e5aa","mcp_get_code":{"code_sha256":"f62b5d3c6983e5aa"}}]}