{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/elu-p1","entry":"elu_p1","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":3,"n_papers":6,"n_buckets":1,"n_distinct_outputs":1,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":0,"by_error":{}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/elu-p1","buckets":[{"bucket":[[2,"float","float32"]],"n_implementations_compared":3,"n_papers":6,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"5746eb64a598b80c","size":3,"n_papers":6,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":3,"torch_reference_conventions_with_this_digest":[],"values":[2.802628517150879,0.21571952104568481,1.476053237915039,0.5011313557624817,0.40541279315948486,0.35683876276016235,0.5154263377189636,1.5728892087936401,1.5188759565353394,1.9591456651687622,0.6721998453140259,0.20838910341262817,0.7493605017662048,3.322108745574951,0.5976296663284302,1.5505285263061523,1.0663954019546509,0.19733047485351562,1.432111382484436,0.6661863327026367,0.9776591062545776,0.30703842639923096,0.5845761299133301,1.983674168586731,0.45130455493927,1.4332743883132935,1.9449777603149414,2.0175182819366455,0.2935270071029663,2.2742180824279785,0.2665424346923828,0.5520541667938232],"values_recorded":32,"members":[{"code_sha256_prefix":"bc4bddefd7540078","path":"fla/layers/delta_net.py","papers":["2402.18668","2406.06484","2603.28743"],"paper_pages":[{"arxiv_id":"2402.18668","page":"/paper/simple-linear-attention-language-models"},{"arxiv_id":"2406.06484","page":"/paper/parallelizing-linear-transformers-with-the"},{"arxiv_id":"2603.28743","page":"/paper/arxiv-2603-28743"}],"arg_sig_recorded":[["x",2,"float","float32"]],"scalar_args":{},"class_bearing":false},{"code_sha256_prefix":"fa08c36ab5558b55","path":"WikiText103/src/utils/AID.py","papers":["2406.01012","2406.06976"],"paper_pages":[{"arxiv_id":"2406.01012","page":"/paper/attention-based-iterative-decomposition-for"},{"arxiv_id":"2406.06976","page":"/paper/discrete-dictionary-based-decomposition-layer"}],"arg_sig_recorded":[["x",2,"float","float32"]],"scalar_args":{},"class_bearing":false},{"code_sha256_prefix":"ee8f82afd691729b","path":"flash-linear-attention/fla/layers/gated_deltaproduct.py","papers":["2502.10297"],"paper_pages":[{"arxiv_id":"2502.10297","page":"/paper/deltaproduct-increasing-the-expressivity-of"}],"arg_sig_recorded":[["x",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]}]}]}