{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/asymmetric-l2-loss","entry":"asymmetric_l2_loss","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":4,"n_papers":8,"n_buckets":1,"n_distinct_outputs":2,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":0,"by_error":{}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/asymmetric-l2-loss","buckets":[{"bucket":[[2,"float","float32"]],"n_implementations_compared":4,"n_papers":8,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"71ba741bd9ee2593","size":3,"n_papers":5,"shape":[],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":3,"torch_reference_conventions_with_this_digest":[],"values":[0.5065228939056396],"values_recorded":1,"members":[{"code_sha256_prefix":"579d77f78c03773e","path":"SRPO.py","papers":["2310.07297","2407.09024","2110.06169","2608.02332"],"paper_pages":[{"arxiv_id":"2310.07297","page":"/paper/score-regularized-policy-optimization-through"},{"arxiv_id":"2407.09024","page":"/paper/aligning-diffusion-behaviors-with-q-functions"},{"arxiv_id":"2110.06169","page":"/paper/offline-reinforcement-learning-with-implicit"},{"arxiv_id":"2608.02332","page":"/paper/arxiv-2608-02332"}],"arg_sig_recorded":[["u",2,"float","float32"]],"scalar_args":{"tau":"0.5"},"class_bearing":false},{"code_sha256_prefix":"b2bdeb779182394b","path":"IQL/iql.py","papers":["2110.06169"],"paper_pages":[{"arxiv_id":"2110.06169","page":"/paper/offline-reinforcement-learning-with-implicit"}],"arg_sig_recorded":[["u",2,"float","float32"]],"scalar_args":{"tau":"0.5"},"class_bearing":false},{"code_sha256_prefix":"e73cbb034680db95","path":"src/iql.py","papers":["2408.07753"],"paper_pages":[{"arxiv_id":"2408.07753","page":"/paper/how-to-solve-contextual-goal-oriented"}],"arg_sig_recorded":[["u",2,"float","float32"]],"scalar_args":{"tau":"0.5"},"class_bearing":false}]},{"output_sha":"bfbd196aab86ae15","size":1,"n_papers":3,"shape":[],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.5162345767021179],"values_recorded":1,"members":[{"code_sha256_prefix":"82a8f299419d4d83","path":"osrl/algorthims/capsiql.py","papers":["2512.02486","2412.08794","2412.18946"],"paper_pages":[{"arxiv_id":"2512.02486","page":"/paper/arxiv-2512-02486"},{"arxiv_id":"2412.08794","page":"/paper/latent-safety-constrained-policy-approach-for"},{"arxiv_id":"2412.18946","page":"/paper/constraint-adaptive-policy-switching-for"}],"arg_sig_recorded":[["u",2,"float","float32"]],"scalar_args":{"tau":"0.25"},"class_bearing":false}]}]}]}