{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/ema","entry":"ema","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":4,"n_papers":7,"n_buckets":2,"n_distinct_outputs":2,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":0,"by_error":{}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/ema-2","buckets":[{"bucket":[[2,"float","float32"],[2,"float","float32"]],"n_implementations_compared":3,"n_papers":6,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"d7bd5ceec8a69dcd","size":3,"n_papers":6,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":1.1920928955078125e-07,"members_with_recorded_values":3,"torch_reference_conventions_with_this_digest":[],"values":[1.8026286363601685,-1.5337762832641602,0.47605323791503906,-0.6908870935440063,-0.9028494954109192,-1.0304712057113647,-0.6627609133720398,0.5728892087936401,0.5188759565353394,0.9591456651687622,-0.3971995711326599,-1.5683482885360718,-0.28853508830070496,2.322108745574951,-0.5147839784622192,0.5505285859107971,0.06639543920755386,-1.6228755712509155,0.4321114420890808,-0.4061858654022217,-0.02259422466158867,-1.180782437324524,-0.5368683338165283,0.983674168586731,-0.7956128120422363,0.43327441811561584,0.9449777603149414,1.0175182819366455,-1.225785732269287,1.274217963218689,-1.3222218751907349,-0.5941091179847717],"values_recorded":32,"members":[{"code_sha256_prefix":"a1073d8f05df7b5a","path":"mine/models/mine.py","papers":["1801.04062","1606.03657","1703.00810","2311.01489"],"paper_pages":[{"arxiv_id":"1801.04062","page":"/paper/mine-mutual-information-neural-estimation"},{"arxiv_id":"1606.03657","page":"/paper/infogan-interpretable-representation-learning"},{"arxiv_id":"1703.00810","page":"/paper/opening-the-black-box-of-deep-neural-networks"},{"arxiv_id":"2311.01489","page":"/paper/invariant-causal-imitation-learning-for-1"}],"arg_sig_recorded":[["mu",2,"float","float32"],["past_ema",2,"float","float32"]],"scalar_args":{"alpha":"0.1"},"class_bearing":false},{"code_sha256_prefix":"62b973828672478f","path":"fairseq/modules/kmeans_attention.py","papers":["2105.14850"],"paper_pages":[{"arxiv_id":"2105.14850","page":"/paper/cascaded-head-colliding-attention"}],"arg_sig_recorded":[["old",2,"float","float32"],["new",2,"float","float32"]],"scalar_args":{"decay":"0.99"},"class_bearing":false},{"code_sha256_prefix":"f70774c0916069a5","path":"group/models/cluster_transformer.py","papers":["2108.12630"],"paper_pages":[{"arxiv_id":"2108.12630","page":"/paper/groupformer-group-activity-recognition-with"}],"arg_sig_recorded":[["old",2,"float","float32"],["new",2,"float","float32"]],"scalar_args":{"decay":"0.9"},"class_bearing":false}]}]},{"bucket":[[2,"float","float32"]],"n_implementations_compared":1,"n_papers":1,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"2465a8f4249dcce1","size":1,"n_papers":1,"shape":[8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.16177573800086975,-0.4093591272830963,0.527328610420227,-0.06555110216140747,-0.7323014736175537,0.33810126781463623,-0.8750224113464355,0.2014583796262741],"values_recorded":8,"members":[{"code_sha256_prefix":"88375ce705173d0e","path":"restkv/llama_model.py","papers":["2605.08840"],"paper_pages":[{"arxiv_id":"2605.08840","page":"/paper/arxiv-2605-08840"}],"arg_sig_recorded":[["input_tensor",2,"float","float32"]],"scalar_args":{"dim":"0","alpha":"None"},"class_bearing":false}]}]}]}