{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/cross-entropy","entry":"cross_entropy","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":6,"n_papers":8,"n_buckets":2,"n_distinct_outputs":5,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":9,"by_error":{"ValueError":6,"IndexError":2,"RuntimeError":1}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":1,"n_papers":1,"recorded_shared_digest_equals_own_fixture_digest":1},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/cross-entropy","buckets":[{"bucket":[[2,"float","float32"],[2,"float","float32"]],"n_implementations_compared":5,"n_papers":7,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"2399fcd7f479a73c","size":2,"n_papers":2,"shape":[],"dtype":"float32","type":"Tensor","finite":false,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[NaN],"values_recorded":1,"members":[{"code_sha256_prefix":"35d4ddeeaaf09552","path":"utils/losses.py","papers":["2401.08552"],"paper_pages":[{"arxiv_id":"2401.08552","page":"/paper/explaining-time-series-via-contrastive-and"}],"arg_sig_recorded":[["proba_pred",2,"float","float32"],["proba_target",2,"float","float32"]],"scalar_args":{},"class_bearing":false},{"code_sha256_prefix":"7671bae0a2497878","path":"mDLAM.py","papers":["1811.04187"],"paper_pages":[{"arxiv_id":"1811.04187","page":"/paper/the-global-convergence-of-the-alternating"}],"arg_sig_recorded":[["label",2,"float","float32"],["prob",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]},{"output_sha":"f8000e4c21127695","size":1,"n_papers":3,"shape":[4],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[-13.750007629394531,-5.2393035888671875,-10.40101432800293,-8.642244338989258],"values_recorded":4,"members":[{"code_sha256_prefix":"63e6e57b587c3aa8","path":"CLIP.py","papers":["2203.03897","2103.00020","2407.08216"],"paper_pages":[{"arxiv_id":"2203.03897","page":"/paper/geodesic-multi-modal-mixup-for-robust-fine"},{"arxiv_id":"2103.00020","page":"/paper/learning-transferable-visual-models-from"},{"arxiv_id":"2407.08216","page":"/paper/multimodal-contrastive-learning-for-spatial"}],"arg_sig_recorded":[["preds",2,"float","float32"],["targets",2,"float","float32"]],"scalar_args":{"reduction":"'none'"},"class_bearing":false}]},{"output_sha":"ba10c57de7deee16","size":1,"n_papers":1,"shape":[],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[-9.508142471313477],"values_recorded":1,"members":[{"code_sha256_prefix":"339e9f8214d3d974","path":"UGformerV2_PyTorch/train_UGformerV2.py","papers":["1909.11855"],"paper_pages":[{"arxiv_id":"1909.11855","page":"/paper/unsupervised-universal-self-attention-network"}],"arg_sig_recorded":[["pred",2,"float","float32"],["soft_targets",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]},{"output_sha":"fbf16c41c2a01696","size":1,"n_papers":1,"shape":[],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[-1.1885178089141846],"values_recorded":1,"members":[{"code_sha256_prefix":"d26b88cdc1bb2436","path":"ood_training.py","papers":["2012.06575"],"paper_pages":[{"arxiv_id":"2012.06575","page":"/paper/entropy-maximization-and-meta-classification"}],"arg_sig_recorded":[["logits",2,"float","float32"],["targets",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]}]},{"bucket":[[1,"float","float64"],[1,"float","float64"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"af5570f5a1810b7a","size":1,"n_papers":1,"shape":[],"dtype":"float64","type":"float64","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.0],"values_recorded":1,"members":[{"code_sha256_prefix":"636e79ded1f9a112","path":"visualize_atari/saliency.py","papers":["1912.12191"],"paper_pages":[{"arxiv_id":"1912.12191","page":"/paper/explain-your-move-understanding-agent-actions-1"}],"arg_sig_recorded":[["L_policy",1,"float","float64"],["l_policy",1,"float","float64"]],"scalar_args":{"L_idx":"2"},"class_bearing":false}]}]}]}