{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/sum-norm","entry":"sum_norm","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":3,"n_papers":6,"n_buckets":1,"n_distinct_outputs":2,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":0,"by_error":{}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/sum-norm","buckets":[{"bucket":[[2,"float","float32"]],"n_implementations_compared":3,"n_papers":6,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"57a439b58dcd2955","size":2,"n_papers":5,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[-0.9154238700866699,0.7788932919502258,-0.2417527735233307,0.35085123777389526,0.4584915339946747,0.5233013033866882,0.33656802773475647,-0.2909287214279175,0.32803043723106384,0.6063664555549622,-0.2511073052883148,-0.9915007948875427,-0.18241024017333984,1.4680240154266357,-0.3254435062408447,0.3480410575866699,-0.029030080884695053,0.7095699310302734,-0.18893209099769592,0.17759665846824646,0.009878873825073242,0.516273558139801,0.2347349375486374,-0.43009188771247864,2.9715747833251953,-1.6182585954666138,-3.5294454097747803,-3.8003807067871094,4.578249454498291,-4.759141445159912,4.938433647155762,2.218968391418457],"values_recorded":32,"members":[{"code_sha256_prefix":"e544fdb6373b9f42","path":"flash-linear-attention/fla/layers/gated_deltaproduct.py","papers":["2502.10297","2402.18668","2603.28743"],"paper_pages":[{"arxiv_id":"2502.10297","page":"/paper/deltaproduct-increasing-the-expressivity-of"},{"arxiv_id":"2402.18668","page":"/paper/simple-linear-attention-language-models"},{"arxiv_id":"2603.28743","page":"/paper/arxiv-2603-28743"}],"arg_sig_recorded":[["x",2,"float","float32"]],"scalar_args":{},"class_bearing":false},{"code_sha256_prefix":"379d7685e2a746e5","path":"WikiText103/src/utils/AID.py","papers":["2406.01012","2406.06976"],"paper_pages":[{"arxiv_id":"2406.01012","page":"/paper/attention-based-iterative-decomposition-for"},{"arxiv_id":"2406.06976","page":"/paper/discrete-dictionary-based-decomposition-layer"}],"arg_sig_recorded":[["x",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]},{"output_sha":"5efe074797531892","size":1,"n_papers":1,"shape":[4,8],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":null,"members_with_recorded_values":1,"torch_reference_conventions_with_this_digest":[],"values":[0.23495234549045563,-0.1999104619026184,0.06204817816615105,-0.09004934132099152,-0.1176762655377388,-0.13431032001972198,-0.08638342469930649,0.07466965913772583,0.07288069278001785,0.13472044467926025,-0.05579017102718353,-0.22028829157352448,-0.04052728787064552,0.32616057991981506,-0.07230593264102936,0.07732658088207245,0.012643168680369854,-0.30903160572052,0.0822836235165596,-0.07734682410955429,-0.004302443005144596,-0.2248472422361374,-0.10223166644573212,0.18731343746185303,-0.10457970201969147,0.05695195868611336,0.12421304732561111,0.13374817371368408,-0.16112397611141205,0.1674901843070984,-0.17380008101463318,-0.07809295505285263],"values_recorded":32,"members":[{"code_sha256_prefix":"1837bde376a0260d","path":"fla/layers/delta_net.py","papers":["2406.06484"],"paper_pages":[{"arxiv_id":"2406.06484","page":"/paper/parallelizing-linear-transformers-with-the"}],"arg_sig_recorded":[["x",2,"float","float32"]],"scalar_args":{},"class_bearing":false}]}]}]}