{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/kl-loss","entry":"kl_loss","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":8,"n_papers":11,"n_buckets":2,"n_distinct_outputs":5,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":2,"by_error":{"IndexError":1,"RuntimeError":1}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/kl-loss","buckets":[{"bucket":[[2,"float","float32"],[2,"float","float32"]],"n_implementations_compared":7,"n_papers":9,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"af5570f5a1810b7a","size":2,"n_papers":2,"shape":[],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":1.792795956134796e-08,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[2.561137080192566e-09],"values_recorded":1,"members":[{"code_sha256_prefix":"9b4fc11504440ef7","path":"ACL_RCS/ImageNet_32/RCS.py","papers":["2302.03857"],"paper_pages":[{"arxiv_id":"2302.03857","page":"/paper/efficient-adversarial-contrastive-learning-1"}],"arg_sig_recorded":[["nat",2,"float","float32"],["adv",2,"float","float32"]],"scalar_args":{"reduction":"'mean'"},"class_bearing":false},{"code_sha256_prefix":"9f77e2bc0dc49ca0","path":"smart_pytorch/loss.py","papers":["1911.03437"],"paper_pages":[{"arxiv_id":"1911.03437","page":"/paper/smart-robust-and-efficient-fine-tuning-for"}],"arg_sig_recorded":[["input",2,"float","float32"],["target",2,"float","float32"]],"scalar_args":{"reduction":"'batchmean'"},"class_bearing":false}]},{"output_sha":"cf3999a9f5564b9c","size":2,"n_papers":2,"shape":[],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[6.661041259765625],"values_recorded":1,"members":[{"code_sha256_prefix":"6a6e7e5aadc14e95","path":"train_vae.py","papers":["2001.04296"],"paper_pages":[{"arxiv_id":"2001.04296","page":"/paper/high-fidelity-synthesis-with-disentangled"}],"arg_sig_recorded":[["mean",2,"float","float32"],["logvar",2,"float","float32"]],"scalar_args":{},"class_bearing":false},{"code_sha256_prefix":"89c4776d4a9fbb8b","path":"training.py","papers":["1605.06432"],"paper_pages":[{"arxiv_id":"1605.06432","page":"/paper/deep-variational-bayes-filters-unsupervised"}],"arg_sig_recorded":[["mu",2,"float","float32"],["log_var",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]},{"output_sha":"e6ad6c9a3a3b7658","size":2,"n_papers":2,"shape":[],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":1.2293457984924316e-07,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[-1.7881393432617188e-07],"values_recorded":1,"members":[{"code_sha256_prefix":"05c4011b1c6139b1","path":"src/learners/baseline/ours.py","papers":["2411.13852"],"paper_pages":[{"arxiv_id":"2411.13852","page":"/paper/dealing-with-synthetic-data-contamination-in"}],"arg_sig_recorded":[["logits_stu",2,"float","float32"],["logits_tea",2,"float","float32"]],"scalar_args":{"temperature":"4.0"},"class_bearing":false},{"code_sha256_prefix":"7ee01bf5efd50d92","path":"dfdg/training/train_student_syn_img.py","papers":["2110.04545"],"paper_pages":[{"arxiv_id":"2110.04545","page":"/paper/towards-data-free-domain-generalization"}],"arg_sig_recorded":[["y",2,"float","float32"],["teacher_scores",2,"float","float32"]],"scalar_args":{"temp":"3","softmax_applied":"False"},"class_bearing":false}]},{"output_sha":"28480b34e5924502","size":1,"n_papers":3,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":false,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[-2.1872682571411133,NaN,-0.5799555778503418,NaN,NaN,NaN,NaN,-0.6473273634910583,-0.6096518039703369,-0.959958553314209,NaN,NaN,NaN,-3.435858726501465,NaN,-0.6316692233085632,-0.1844712197780609,NaN,-0.5492827296257019,NaN,NaN,NaN,NaN,-0.9837967157363892,NaN,-0.5501005053520203,-0.9464529156684875,-1.017662525177002,NaN,-1.314836859703064,NaN,NaN],"values_recorded":32,"members":[{"code_sha256_prefix":"9960f247d801fa2a","path":"mobilenet_v2_rslad_cifar10.py","papers":["1706.06083","1905.09747","2306.16170"],"paper_pages":[{"arxiv_id":"1706.06083","page":"/paper/towards-deep-learning-models-resistant-to"},{"arxiv_id":"1905.09747","page":"/paper/190509747"},{"arxiv_id":"2306.16170","page":"/paper/mitigating-the-accuracy-robustness-trade-off"}],"arg_sig_recorded":[["a",2,"float","float32"],["b",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]}]},{"bucket":[[2,"float","float32"],[2,"float","float32"],[2,"float","float32"],[2,"float","float32"]],"n_implementations_compared":1,"n_papers":2,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"5341e6b2646979a7","size":1,"n_papers":2,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0],"values_recorded":32,"members":[{"code_sha256_prefix":"1cb058c996e65e63","path":"models/kovae.py","papers":["2310.02619","2605.17804"],"paper_pages":[{"arxiv_id":"2310.02619","page":"/paper/generative-modeling-of-regular-and-irregular"},{"arxiv_id":"2605.17804","page":"/paper/arxiv-2605-17804"}],"arg_sig_recorded":[["z_post_mean",2,"float","float32"],["z_post_logvar",2,"float","float32"],["z_prior_mean",2,"float","float32"],["z_prior_logvar",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]}]}]}