{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/gumbel-softmax","entry":"gumbel_softmax","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":19,"n_papers":20,"n_buckets":2,"n_distinct_outputs":12,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":0,"by_error":{}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/gumbel-softmax","buckets":[{"bucket":[[2,"float","float32"]],"n_implementations_compared":18,"n_papers":19,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"4b71b65a282dc2bb","size":4,"n_papers":4,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":1.862645149230957e-08,"members_with_recorded_values":4,"torch_reference_conventions_with_this_digest":[],"values":[0.220077782869339,0.0019655548967421055,0.0012939530424773693,0.00017989642219617963,0.0003469175717327744,0.001801609294489026,0.001533946837298572,0.7728002667427063,0.058907076716423035,0.4178870618343353,0.005254834890365601,0.0006732303882017732,0.0005009144078940153,0.4238051772117615,0.0030712641309946775,0.0899004265666008,0.0207397248595953,0.0018409527838230133,0.0016749952919781208,0.0006529040983878076,0.015313185751438141,0.028255123645067215,0.0009430024074390531,0.9305800795555115,9.28233566810377e-05,0.002330946736037731,0.9716299772262573,0.00023893694742582738,1.042726307787234e-05,0.004534349776804447,1.7346590539091267e-05,0.021145295351743698],"values_recorded":32,"members":[{"code_sha256_prefix":"8a5f974f305016c7","path":"models_dy.py","papers":["2007.00631"],"paper_pages":[{"arxiv_id":"2007.00631","page":"/paper/causal-discovery-in-physical-systems-from"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"0.5","hard":"False"},"class_bearing":false},{"code_sha256_prefix":"a71884d37a633b0f","path":"model/pytorch/model.py","papers":["2101.06861"],"paper_pages":[{"arxiv_id":"2101.06861","page":"/paper/discrete-graph-structure-learning-for-1"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"0.5","hard":"False","eps":"1e-10"},"class_bearing":false},{"code_sha256_prefix":"ad6724b21272b7b3","path":"model_vmtl.py","papers":["2111.05323"],"paper_pages":[{"arxiv_id":"2111.05323","page":"/paper/variational-multi-task-learning-with-gumbel"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"0.5","hard":"False"},"class_bearing":false},{"code_sha256_prefix":"ae5d3a400d0e7968","path":"gumbel_sigmoid_softmax.py","papers":["1611.01144"],"paper_pages":[{"arxiv_id":"1611.01144","page":"/paper/categorical-reparameterization-with-gumbel"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"0.5","hard":"False"},"class_bearing":false}]},{"output_sha":"ba691d412b40125b","size":3,"n_papers":4,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":3,"torch_reference_conventions_with_this_digest":[],"values":[0.0,0.0,0.0,0.0,0.0,0.0,0.0,1.0,0.0,0.0,0.0,0.0,0.0,1.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,1.0,0.0,0.0,1.0,0.0,0.0,0.0,0.0,0.0],"values_recorded":32,"members":[{"code_sha256_prefix":"4763872bbaf3ab60","path":"vqapc_model.py","papers":["1904.03240","2005.08392"],"paper_pages":[{"arxiv_id":"1904.03240","page":"/paper/an-unsupervised-autoregressive-model-for"},{"arxiv_id":"2005.08392","page":"/paper/vector-quantized-autoregressive-predictive"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"0.5"},"class_bearing":false},{"code_sha256_prefix":"447344e558aec603","path":"models/LSINet.py","papers":["2602.01585"],"paper_pages":[{"arxiv_id":"2602.01585","page":"/paper/arxiv-2602-01585"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"0.5","hard":"True","eps":"1e-10","device":"None"},"class_bearing":false},{"code_sha256_prefix":"c8519676aac0e734","path":"model/cross_modal_attention.py","papers":["2308.15512"],"paper_pages":[{"arxiv_id":"2308.15512","page":"/paper/shatter-and-gather-learning-referring-image"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"dim":"1","tau":"1.0"},"class_bearing":false}]},{"output_sha":"2ddd2140e02ea79c","size":2,"n_papers":2,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[0.3041975498199463,0.028748169541358948,0.023325279355049133,0.008697188459336758,0.012077602557837963,0.027523133903741837,0.0253964364528656,0.5700346231460571,0.12037739902734756,0.320620059967041,0.03595346584916115,0.012868949212133884,0.011100511066615582,0.32288238406181335,0.02748652547597885,0.14871065318584442,0.09347780048847198,0.02785019762814045,0.026565242558717728,0.016585618257522583,0.08032297343015671,0.10910774767398834,0.019932575523853302,0.6261578798294067,0.007531470153480768,0.03774133697152138,0.770551323890686,0.012083500623703003,0.002524272771552205,0.05263912305235863,0.003255803370848298,0.11367318034172058],"values_recorded":32,"members":[{"code_sha256_prefix":"57d033c0c861f3d0","path":"maddpg-pytorch/algorithms/maddpg.py","papers":["1706.02275"],"paper_pages":[{"arxiv_id":"1706.02275","page":"/paper/multi-agent-actor-critic-for-mixed"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"1.0","hard":"False"},"class_bearing":false},{"code_sha256_prefix":"706773b496bb4a28","path":"gumbel_sigmoid_softmax.py","papers":["1602.02830"],"paper_pages":[{"arxiv_id":"1602.02830","page":"/paper/binarized-neural-networks-training-deep"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"1.0","hard":"False"},"class_bearing":false}]},{"output_sha":"8c514d1427822e8c","size":2,"n_papers":2,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[0.3041975498199463,0.028748171404004097,0.023325283080339432,0.008697187528014183,0.012077601626515388,0.027523139491677284,0.02539643831551075,0.5700345635414124,0.12037741392850876,0.320620059967041,0.03595346584916115,0.012868942692875862,0.011100515723228455,0.32288238406181335,0.02748652920126915,0.14871065318584442,0.09347778558731079,0.02785019762814045,0.02656523510813713,0.016585614532232285,0.08032297343015671,0.10910773277282715,0.019932575523853302,0.6261578798294067,0.00753146642819047,0.03774133697152138,0.770551323890686,0.012083495035767555,0.002524272771552205,0.05263912305235863,0.003255803370848298,0.11367318034172058],"values_recorded":32,"members":[{"code_sha256_prefix":"1824fbfcfee2e185","path":"model_search.py","papers":["2106.06575"],"paper_pages":[{"arxiv_id":"2106.06575","page":"/paper/auto-nba-efficient-and-effective-search-over"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"1.0","hard":"False"},"class_bearing":false},{"code_sha256_prefix":"5109085229be38d2","path":"model_search.py","papers":["2210.13361"],"paper_pages":[{"arxiv_id":"2210.13361","page":"/paper/nasa-neural-architecture-search-and"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"temperature":"1.0","hard":"False"},"class_bearing":false}]},{"output_sha":"d518dad2968195bc","size":1,"n_papers":2,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.304197758436203,0.028748180717229843,0.023325301706790924,0.008697192184627056,0.012077610939741135,0.027523159980773926,0.025396453216671944,0.5700343251228333,0.12037740647792816,0.320620059967041,0.03595346957445145,0.012868950143456459,0.01110051479190588,0.32288241386413574,0.0274865310639143,0.14871065318584442,0.09347784519195557,0.027850208804011345,0.02656526491045952,0.01658562943339348,0.08032304048538208,0.10910765826702118,0.019932586699724197,0.626157820224762,0.007531485520303249,0.037741415202617645,0.7705510854721069,0.012083525769412518,0.0025242792908102274,0.05263924226164818,0.0032558117527514696,0.11367322504520416],"values_recorded":32,"members":[{"code_sha256_prefix":"7f29142583965575","path":"model.py","papers":["2303.17056","2409.19865"],"paper_pages":[{"arxiv_id":"2303.17056","page":"/paper/audio-visual-grouping-network-for-sound"},{"arxiv_id":"2409.19865","page":"/paper/tokenbinder-text-video-retrieval-with-one-to"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"tau":"1.0","hard":"False","dim":"-1"},"class_bearing":false}]},{"output_sha":"08b19d1cfcec511a","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.5474895238876343,-1.741103172302246,0.9621145129203796,-1.6211273670196533,-0.9379322528839111,-1.4790780544281006,0.9630473256111145,1.4110889434814453,0.657335638999939,2.0379295349121094,0.10852783918380737,0.06868958473205566,0.5349061489105225,2.3498804569244385,-0.09545427560806274,1.093104600906372,0.7145007848739624,-0.15304744243621826,0.34250959753990173,-0.8357131481170654,1.4205399751663208,-1.3023765087127686,1.6152608394622803,1.2823272943496704,0.5661468505859375,0.6353961229324341,1.0664796829223633,0.8816481828689575,-1.5391120910644531,2.627755641937256,-1.8291385173797607,-0.5899494886398315],"values_recorded":32,"members":[{"code_sha256_prefix":"7e5297fa7f8589bf","path":"model/RGSL.py","papers":["2210.06126"],"paper_pages":[{"arxiv_id":"2210.06126","page":"/paper/regularized-graph-structure-learning-with"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"tau":"1.0","hard":"False","eps":"1e-10","dim":"-1"},"class_bearing":false}]},{"output_sha":"3fcfa9b79d413d05","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.1939879208803177,4.21536672234879e-09,0.0011294428259134293,7.149050795796086e-11,2.1064956723382267e-10,1.4120355995572709e-09,5.762445520751669e-10,0.8048827052116394,0.06033804267644882,0.6063331365585327,1.2489593803621801e-09,1.6649448486560914e-09,9.580063131675587e-11,0.23602603375911713,9.235053277656391e-10,0.09730273485183716,0.0006347914459183812,3.748665378111582e-09,0.0010449605761095881,1.1664874621786225e-10,1.2703184060214312e-09,2.3764801682091274e-08,2.1880279532648927e-10,0.9983201622962952,3.455848454625432e-11,0.0013949483400210738,0.9939940571784973,0.00024513175594620407,9.177228421641814e-12,0.004365841392427683,1.851476437442212e-11,5.261228785968797e-09],"values_recorded":32,"members":[{"code_sha256_prefix":"3cb0d3bdcd72602a","path":"mixed_operation.py","papers":["1806.09055"],"paper_pages":[{"arxiv_id":"1806.09055","page":"/paper/darts-differentiable-architecture-search"}],"arg_sig_recorded":[["alphas",2,"float","float32"]],"scalar_args":{"temp":"0.5","epsilon":"0.0001"},"class_bearing":false}]},{"output_sha":"5a943745493c4d54","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[-1.942280650138855,-4.230873107910156,-1.5276556015014648,-4.110897541046143,-3.4277024269104004,-3.96884822845459,-1.5267229080200195,-1.078681230545044,-2.6703648567199707,-1.2897710800170898,-3.219172716140747,-3.2590110301971436,-2.7927944660186768,-0.9778201580047607,-3.423154830932617,-2.234596014022827,-2.1640658378601074,-3.031614065170288,-2.5360569953918457,-3.7142796516418457,-1.458026647567749,-4.180943012237549,-1.2633057832717896,-1.5962393283843994,-2.6010732650756836,-2.5318241119384766,-2.100740432739258,-2.285572052001953,-4.706332206726074,-0.5394644737243652,-4.996358394622803,-3.757169723510742],"values_recorded":32,"members":[{"code_sha256_prefix":"e6d0d3fd5a52390a","path":"src/models/dams.py","papers":["2109.04080"],"paper_pages":[{"arxiv_id":"2109.04080","page":"/paper/low-resource-dialogue-summarization-with"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"tau":"1.0","hard":"False","log_mode":"True","dim":"-1"},"class_bearing":false}]},{"output_sha":"8de8fd95c34475b9","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.6061398983001709,0.08182819932699203,0.012006753124296665,0.13480830192565918,0.10525322705507278,0.05754581838846207,0.3494420349597931,0.34822458028793335,0.14965417981147766,0.5693904757499695,0.011546919122338295,0.12445363402366638,0.06035654991865158,0.4211985170841217,0.23596572875976562,0.0566796250641346,0.20783919095993042,0.08845511823892593,0.015258586965501308,0.2868608832359314,0.78108149766922,0.2545500695705414,0.30603253841400146,0.4268191158771515,0.03636676445603371,0.26032620668411255,0.9611877202987671,0.45387718081474304,0.053308796137571335,0.2667056918144226,0.10855963081121445,0.16827671229839325],"values_recorded":32,"members":[{"code_sha256_prefix":"422ac583248740e3","path":"model/GroupNet_nba.py","papers":["2204.08770"],"paper_pages":[{"arxiv_id":"2204.08770","page":"/paper/groupnet-multiscale-hypergraph-neural"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"tau":"1.0","hard":"False","eps":"1e-10"},"class_bearing":false}]},{"output_sha":"9afd589f4243e47f","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.8326823711395264,0.0012805670266970992,0.045749861747026443,0.004596929997205734,0.003351898631080985,0.003141004592180252,0.005991474259644747,0.1032058373093605,0.02739463932812214,0.07360830157995224,0.00413600355386734,0.00040909042581915855,0.003976012114435434,0.8570509552955627,0.0031720309052616358,0.030253024771809578,0.08583275973796844,0.003220498561859131,0.12890370190143585,0.025943392887711525,0.07094363868236542,0.009378341026604176,0.021273290738463402,0.6545043587684631,0.006073881406337023,0.07657939195632935,0.35165974497795105,0.17454524338245392,0.0022502904757857323,0.371878445148468,0.0019904705695807934,0.01502254232764244],"values_recorded":32,"members":[{"code_sha256_prefix":"36c38bfa52ade5c4","path":"gnia.py","papers":["2108.13049"],"paper_pages":[{"arxiv_id":"2108.13049","page":"/paper/single-node-injection-attack-against-graph"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"tau":"0.5","random_flag":"True","eps":"0.1","dim":"-1"},"class_bearing":false}]},{"output_sha":"f2dd12d82e68fd79","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.14337657392024994,0.01453968696296215,0.2170438915491104,0.016393054276704788,0.03246143460273743,0.018895182758569717,0.21724645793437958,0.34004366397857666,0.0692269504070282,0.27533379197120667,0.03998812288045883,0.03842638060450554,0.06124981492757797,0.3761301040649414,0.03260939568281174,0.10703536123037338,0.11485718190670013,0.048237718641757965,0.07917798310518265,0.024372989311814308,0.23269502818584442,0.015284085646271706,0.28271788358688354,0.20265722274780273,0.07419390976428986,0.07951385527849197,0.1223657950758934,0.10171587020158768,0.0090378662571311,0.5830604434013367,0.006762529257684946,0.023349735885858536],"values_recorded":32,"members":[{"code_sha256_prefix":"d599e623fb8e126d","path":"cnn/model_search.py","papers":["1806.09055"],"paper_pages":[{"arxiv_id":"1806.09055","page":"/paper/darts-differentiable-architecture-search"}],"arg_sig_recorded":[["logits",2,"float","float32"]],"scalar_args":{"tau":"1.0","hard":"False","eps":"1e-10","dim":"-1"},"class_bearing":false}]}]},{"bucket":[[2,"float","float32"],[2,"float","float32"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"1d026df1ea20fff7","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":false,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[NaN,NaN,0.6814374923706055,NaN,NaN,NaN,NaN,0.8498054146766663,0.7665320038795471,0.9997336268424988,NaN,NaN,NaN,NaN,NaN,0.8185672163963318,0.005742745939642191,NaN,0.5787732005119324,NaN,NaN,NaN,NaN,0.9999614953994751,NaN,0.5816476941108704,0.9994879961013794,NaN,NaN,NaN,NaN,NaN],"values_recorded":32,"members":[{"code_sha256_prefix":"ef0219182bdf035d","path":"models/sparsefunc.py","papers":["2011.07439"],"paper_pages":[{"arxiv_id":"2011.07439","page":"/paper/efficient-variational-inference-for-sparse-1"}],"arg_sig_recorded":[["logits",2,"float","float32"],["U",2,"float","float32"]],"scalar_args":{"temperature":"0.5","hard":"False","eps":"1e-20"},"class_bearing":false}]}]}]}