{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/masked-softmax","entry":"masked_softmax","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":10,"n_papers":10,"n_buckets":6,"n_distinct_outputs":10,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":3,"by_error":{"RuntimeError":2,"IndexError":1}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/masked-softmax","buckets":[{"bucket":[[2,"float","float32"],[2,"float","float32"]],"n_implementations_compared":5,"n_papers":5,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"564451c2ada11c49","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[1.8908239603042603,-0.6560615301132202,0.024299433454871178,-0.04531318321824074,-0.08301237970590591,-0.12126024812459946,-0.04184461757540703,0.032368674874305725,0.0013744303723797202,0.004870252218097448,-0.0009411510545760393,-0.03713785111904144,-0.0006345821311697364,1.032318353652954,-0.0013578361831605434,0.0015084802871569991,-0.0026253426913172007,0.88967365026474,-0.020503148436546326,0.018858661875128746,0.0008899227832444012,0.18742042779922485,0.028195291757583618,-0.10190939903259277,0.45557111501693726,-0.15894168615341187,-0.7017545104026794,-0.871228814125061,1.6745959520339966,-1.9648255109786987,2.3094825744628906,0.25710076093673706],"values_recorded":32,"members":[{"code_sha256_prefix":"590b22f923c9c19f","path":"Model_code/models.py","papers":["2211.17046"],"paper_pages":[{"arxiv_id":"2211.17046","page":"/paper/raft-rationale-adaptor-for-few-shot-abusive"}],"arg_sig_recorded":[["vec",2,"float","float32"],["mask",2,"float","float32"]],"scalar_args":{"dim":"1"},"class_bearing":false}]},{"output_sha":"cc364006c12c25ca","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[1.8908236026763916,-0.6560613512992859,0.02429942786693573,-0.04531317204236984,-0.08301236480474472,-0.12126020342111588,-0.04184461012482643,0.03236866742372513,0.0013744303723797202,0.004870252683758736,-0.0009411510545760393,-0.037137847393751144,-0.0006345821311697364,1.032318353652954,-0.0013578361831605434,0.001508480403572321,-0.0026253422256559134,0.8896735310554504,-0.020503146573901176,0.018858660012483597,0.0008899227250367403,0.18742042779922485,0.02819528803229332,-0.10190939903259277,0.45557114481925964,-0.15894170105457306,-0.701754629611969,-0.8712288737297058,1.6745961904525757,-1.9648257493972778,2.3094828128814697,0.25710079073905945],"values_recorded":32,"members":[{"code_sha256_prefix":"971291e2c5ef552b","path":"excord.py","papers":["2106.11575"],"paper_pages":[{"arxiv_id":"2106.11575","page":"/paper/learn-to-resolve-conversational-dependency-a"}],"arg_sig_recorded":[["vector",2,"float","float32"],["mask",2,"float","float32"]],"scalar_args":{"dim":"-1","memory_efficient":"False","mask_fill_value":"-1e+32"},"class_bearing":false}]},{"output_sha":"dec3904e83f884ce","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125,0.125],"values_recorded":32,"members":[{"code_sha256_prefix":"42fb56ded96dd10a","path":"model/seq2seqPTR/src/ptrbert.py","papers":["1609.08144"],"paper_pages":[{"arxiv_id":"1609.08144","page":"/paper/googles-neural-machine-translation-system"}],"arg_sig_recorded":[["vector",2,"float","float32"],["mask",2,"float","float32"]],"scalar_args":{"dim":"-1","mask_fill_value":"-1e+32"},"class_bearing":false}]},{"output_sha":"f2360ef3a23d8dac","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[1.890821933746338,-0.6560608148574829,0.02429940737783909,-0.04531313478946686,-0.08301229774951935,-0.12126011401414871,-0.04184457287192345,0.03236864134669304,0.001374429790303111,0.004870249889791012,-0.0009411507053300738,-0.03713783249258995,-0.0006345818401314318,1.032317876815796,-0.0013578356010839343,0.0015084795886650681,-0.002625343855470419,0.8896741271018982,-0.02050315961241722,0.018858671188354492,0.0008899232489056885,0.18742051720619202,0.028195304796099663,-0.10190945118665695,0.4555719196796417,-0.15894196927547455,-0.7017557621002197,-0.8712303638458252,1.6745989322662354,-1.9648290872573853,2.3094866275787354,0.2571012079715729],"values_recorded":32,"members":[{"code_sha256_prefix":"e5002de1b930b693","path":"alfworld/agents/modules/model.py","papers":["2010.03768"],"paper_pages":[{"arxiv_id":"2010.03768","page":"/paper/alfworld-aligning-text-and-embodied"}],"arg_sig_recorded":[["x",2,"float","float32"],["m",2,"float","float32"]],"scalar_args":{"axis":"-1"},"class_bearing":false}]},{"output_sha":"feda954aa726c243","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.9972796440124512,-0.030178004875779152,0.06989431381225586,-0.031578946858644485,-0.033385030925273895,-0.033538755029439926,-0.031157495453953743,0.09266425669193268,0.03241967409849167,0.0930757075548172,-0.00992904044687748,-0.012153955176472664,-0.00804062094539404,0.8805655837059021,-0.011440825648605824,0.03550352901220322,0.03414645791053772,-0.15411736071109772,0.32035505771636963,-0.13022452592849731,-0.010630584321916103,-0.17447564005851746,-0.15103620290756226,1.265982747077942,-0.03958183526992798,0.07366414368152618,0.2680060863494873,0.310291051864624,-0.039663124829530716,0.502289354801178,-0.038850363343954086,-0.036155328154563904],"values_recorded":32,"members":[{"code_sha256_prefix":"c8b051f5575732d3","path":"model/basic.py","papers":["1611.01144"],"paper_pages":[{"arxiv_id":"1611.01144","page":"/paper/categorical-reparameterization-with-gumbel"}],"arg_sig_recorded":[["logits",2,"float","float32"],["mask",2,"float","float32"]],"scalar_args":{"dim":"-1"},"class_bearing":false}]}]},{"bucket":[[1,"float","float64"],[1,"int","int64"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"6724e8d7aae67a28","size":1,"n_papers":1,"shape":[8],"dtype":"float64","type":"ndarray","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.023778083257505685,0.03435122090996154,0.21609764796660647,0.0,0.0246124039344987,0.06191020356256833,0.011521478661455458,0.049161950146330714],"values_recorded":8,"members":[{"code_sha256_prefix":"fd360cf9f53416f8","path":"private/grammar.py","papers":["2203.08031"],"paper_pages":[{"arxiv_id":"2203.08031","page":"/paper/data-efficient-graph-grammar-learning-for-1"}],"arg_sig_recorded":[["logit",1,"float","float64"],["mask",1,"int","int64"]],"scalar_args":{},"class_bearing":false}]}]},{"bucket":[[2,"float","float32"],[1,"int","int64"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"7ebb8c96de1fe0e5","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":false,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.9656568765640259,0.03434319049119949,0.0,0.0,0.0,0.0,0.0,0.0,1.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.38077831268310547,0.07031227648258209,0.5489093661308289,0.0,0.0,0.0,0.0,0.0,NaN,NaN,NaN,NaN,NaN,NaN,NaN,NaN],"values_recorded":32,"members":[{"code_sha256_prefix":"79d2a0642224e83f","path":"text_rationale_attention/attention_model.py","papers":["2104.14403"],"paper_pages":[{"arxiv_id":"2104.14403","page":"/paper/do-feature-attribution-methods-correctly"}],"arg_sig_recorded":[["attn_odds",2,"float","float32"],["lens",1,"int","int64"]],"scalar_args":{},"class_bearing":false}]}]},{"bucket":[[2,"float","float32"],[2,"bool","bool"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"728c40cf20662b28","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.7354134321212769,0.02615467645227909,0.19516736268997192,0.0,0.0,0.04326452314853668,0.0,0.0,0.0,0.0,0.0,0.015451449900865555,0.05556291341781616,0.7560895085334778,0.04431251436471939,0.128583624958992,0.0,0.0,0.6117574572563171,0.0,0.38824254274368286,0.0,0.0,0.0,0.049154166132211685,0.16798065602779388,0.0,0.3012958765029907,0.03196970373392105,0.38947218656539917,0.0,0.0601273812353611],"values_recorded":32,"members":[{"code_sha256_prefix":"27a750666daf5a7f","path":"modules/named_models.py","papers":["2207.06652"],"paper_pages":[{"arxiv_id":"2207.06652","page":"/paper/every-preference-changes-differently-neural"}],"arg_sig_recorded":[["vec",2,"float","float32"],["mask",2,"bool","bool"]],"scalar_args":{"dim":"-1","epsilon":"0.001"},"class_bearing":false}]}]},{"bucket":[[2,"float","float32"],[2,"int","int64"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"053f560feb52191b","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[-0.9010061025619507,3.0679216384887695,-0.4125528931617737,0.006140489596873522,0.004967625718563795,2.062612771987915,-0.6627609133720398,0.021729717031121254,0.0005264815408736467,0.9591456651687622,0.00021063793974462897,3.136704921722412,0.5774657726287842,-3.647444009780884,1.0297685861587524,-0.5486438274383545,0.06639543920755386,-1.6228755712509155,0.0465029776096344,-0.4061858654022217,0.2153615653514862,-1.180782437324524,0.05894807353615761,0.26967012882232666,1.5938800573349,-0.7606090307235718,0.02476455271244049,-1.4237275123596191,2.4523017406463623,-1.0280488729476929,0.0025656544603407383,0.5999762415885925],"values_recorded":32,"members":[{"code_sha256_prefix":"d2b7df2f7dde0879","path":"hxetda/hxetda.py","papers":["2312.02266"],"paper_pages":[{"arxiv_id":"2312.02266","page":"/paper/hierarchical-cross-entropy-loss-for"}],"arg_sig_recorded":[["vec",2,"float","float32"],["mask",2,"int","int64"]],"scalar_args":{"dim":"1","epsilon":"1e-10"},"class_bearing":false}]}]},{"bucket":[[3,"float","float32"],[3,"bool","bool"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"d189fa9dc97b0b5f","size":1,"n_papers":1,"shape":[2,4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.0,0.8134716153144836,0.04293353855609894,0.0,0.0,0.03600646182894707,0.10758833587169647,0.0,0.03183837980031967,0.0,0.11569271236658096,0.0,0.6242156624794006,0.22825327515602112,0.0,0.0,0.0,0.0,0.4262489676475525,0.0,0.2629907727241516,0.0,0.3107602000236511,0.0,0.0,0.3906205892562866,0.07292017340660095,0.0,0.0,0.5364592671394348,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,1.0,0.0,0.0,0.0,0.0773337110877037,0.4103460907936096,0.13299115002155304,0.0,0.3793290853500366,0.0,0.1413600891828537,0.32951614260673523,0.0,0.2642344534397125,0.18816345930099487,0.0767258033156395,0.0,0.025901595130562782,0.293323814868927,0.0,0.05164789780974388,0.6032469868659973,0.0,0.02587970718741417,0.0],"values_recorded":64,"members":[{"code_sha256_prefix":"fbe8aa7e071baf62","path":"model/modules/Attention.py","papers":["1902.10186"],"paper_pages":[{"arxiv_id":"1902.10186","page":"/paper/attention-is-not-explanation"}],"arg_sig_recorded":[["attn_odds",3,"float","float32"],["masks",3,"bool","bool"]],"scalar_args":{},"class_bearing":false}]}]}]}