{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/census/unpad-image","entry":"unpad_image","source":"Syntology differential census (groundwork/55, run v2_2026-09-22), per sample; not an archive number","census_date":"2026-09-22","battery_sha256":["b986f7e04d794a0d88ad4c5f32cf63ec3590b5deff0192150737bc5f1c0b4677"],"runner_sha256":["5a452d0e7c0da5b80771d1be2afe3572e253568cf5c5d59a0d08ebd663d00808"],"bucket_key":"positional (rank, kind, dtype) of each array argument; the argument name is not part of the key because the harness draws the shared array from (rank, kind) and casts it to the dtype, whatever the name","bucket_fields":["rank","kind","dtype"],"claim":"Implementations sharing this entry name were each run on one shared input fixed by the positional (rank, kind, dtype) of their array arguments (the bucket). A cluster is the set whose recorded output digest (sha256 of the output rounded to 6 decimals) is identical. Identical values to six decimals on the shared input are agreement on those inputs, not a statement about the whole domain and not a substitution claim.","n_implementations_compared":4,"n_papers":34,"n_buckets":1,"n_distinct_outputs":2,"n_class_bearing":0,"not_compared":{"not_run_on_shared_input":{"n":0,"by_error":{}},"no_array_argument_ran_on_own_fixture_arguments_only":{"n":0,"n_papers":0,"recorded_shared_digest_equals_own_fixture_digest":0},"output_not_digested_non_numeric":{"n":0,"by_type":{}}},"withdrawn_excluded":0,"code_page":"/code/unpad-image","buckets":[{"bucket":[[3,"float","float32"]],"n_implementations_compared":4,"n_papers":34,"n_not_digested":0,"not_digested_by_type":{},"clusters":[{"output_sha":"646bdac4d0d67392","size":2,"n_papers":18,"shape":[2,4,4],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[-1.2588046789169312,-0.2582058608531952,0.15099599957466125,-1.434759497642517,-0.848949134349823,-0.7722031474113464,0.8366092443466187,-0.16943085193634033,-0.08752579241991043,0.7744670510292053,-0.5704304575920105,1.1243221759796143,-0.7546214461326599,-0.30625444650650024,-1.0338436365127563,1.2410037517547607,-1.1814329624176025,-0.8039284944534302,0.9990944862365723,0.6304919123649597,-1.0145355463027954,-0.6859840750694275,0.9828868508338928,-0.14383146166801453,0.7487973570823669,-0.9978419542312622,0.5280088782310486,0.18848304450511932,1.6535718441009521,-0.581495463848114,1.8763816356658936,1.2408666610717773],"values_recorded":32,"members":[{"code_sha256_prefix":"7606525af238fb64","path":"llava/model/llava_arch.py","papers":["2304.08485","2406.20092","2504.02438","2407.07895","2411.14432","2406.08487","2406.12275","2408.15998","2412.03248","2504.01328","2410.13360","2501.14818","2503.08689","2411.11706","2604.13565","2507.00505"],"paper_pages":[{"arxiv_id":"2304.08485","page":"/paper/visual-instruction-tuning-1"},{"arxiv_id":"2406.20092","page":"/paper/llavolta-efficient-multi-modal-models-via"},{"arxiv_id":"2504.02438","page":"/paper/scaling-video-language-models-to-10k-frames"},{"arxiv_id":"2407.07895","page":"/paper/llava-next-interleave-tackling-multi-image"},{"arxiv_id":"2411.14432","page":"/paper/insight-v-exploring-long-chain-visual"},{"arxiv_id":"2406.08487","page":"/paper/beyond-llava-hd-diving-into-high-resolution"},{"arxiv_id":"2406.12275","page":"/paper/voco-llama-towards-vision-compression-with"},{"arxiv_id":"2408.15998","page":"/paper/eagle-exploring-the-design-space-for"},{"arxiv_id":"2412.03248","page":"/paper/aim-adaptive-inference-of-multi-modal-llms"},{"arxiv_id":"2504.01328","page":"/paper/slow-fast-architecture-for-video-multi-modal"},{"arxiv_id":"2410.13360","page":"/paper/remember-retrieve-and-generate-understanding"},{"arxiv_id":"2501.14818","page":"/paper/eagle-2-building-post-training-data"},{"arxiv_id":"2503.08689","page":"/paper/quota-query-oriented-token-assignment-via-cot"},{"arxiv_id":"2411.11706","page":"/paper/mc-llava-multi-concept-personalized-vision"},{"arxiv_id":"2604.13565","page":"/paper/arxiv-2604-13565"},{"arxiv_id":"2507.00505","page":"/paper/llava-sp-enhancing-visual-representation-with"}],"arg_sig_recorded":[["tensor",3,"float","float32"]],"scalar_args":{"original_size":"(4, 5)"},"class_bearing":false},{"code_sha256_prefix":"83d2977826140fdc","path":"cambrian/model/cambrian_arch.py","papers":["2406.16860","2410.17434"],"paper_pages":[{"arxiv_id":"2406.16860","page":"/paper/cambrian-1-a-fully-open-vision-centric"},{"arxiv_id":"2410.17434","page":"/paper/longvu-spatiotemporal-adaptive-compression"}],"arg_sig_recorded":[["tensor",3,"float","float32"]],"scalar_args":{"original_size":"(4, 5)"},"class_bearing":false}]},{"output_sha":"9dac7894331fbc99","size":2,"n_papers":17,"shape":[2,4,6],"dtype":"float32","type":"Tensor","finite":true,"max_difference_among_recorded_values":0.0,"members_with_recorded_values":2,"torch_reference_conventions_with_this_digest":[],"values":[1.682853102684021,-1.2588046789169312,-0.2582058608531952,0.15099599957466125,-1.434759497642517,-0.34014570713043213,0.011872420087456703,-0.848949134349823,-0.7722031474113464,0.8366092443466187,-0.16943085193634033,0.15168990194797516,-0.7452988624572754,-0.08752579241991043,0.7744670510292053,-0.5704304575920105,1.1243221759796143,-0.40352779626846313,0.9237498044967651,-0.7546214461326599,-0.30625444650650024,-1.0338436365127563,1.2410037517547607,-0.9114314913749695,1.8682522773742676,-1.1814329624176025,-0.8039284944534302,0.9990944862365723,0.6304919123649597,0.044029541313648224,-0.8975064754486084,-1.0145355463027954,-0.6859840750694275,0.9828868508338928,-0.14383146166801453,0.38964250683784485,-0.0975174680352211,0.7487973570823669,-0.9978419542312622,0.5280088782310486,0.18848304450511932,-0.7085897922515869,1.155332088470459,1.6535718441009521,-0.581495463848114,1.8763816356658936,1.2408666610717773,-1.2724858522415161],"values_recorded":48,"members":[{"code_sha256_prefix":"55c32993da87759b","path":"llava/model/llava_arch.py","papers":["2304.08485","2411.00394","2411.03312","2412.00876","2502.13508","2504.13169","2505.19235","2310.03744","2309.09958","2407.15841","2412.01818","2503.15621","2410.22313","2412.02158","2410.02080","2602.23699"],"paper_pages":[{"arxiv_id":"2304.08485","page":"/paper/visual-instruction-tuning-1"},{"arxiv_id":"2411.00394","page":"/paper/right-this-way-can-vlms-guide-us-to-see-more"},{"arxiv_id":"2411.03312","page":"/paper/inference-optimal-vlms-need-only-one-visual"},{"arxiv_id":"2412.00876","page":"/paper/dynamic-llava-efficient-multimodal-large"},{"arxiv_id":"2502.13508","page":"/paper/vlas-vision-language-action-model-with-speech"},{"arxiv_id":"2504.13169","page":"/paper/generate-but-verify-reducing-hallucination-in"},{"arxiv_id":"2505.19235","page":"/paper/corematching-a-co-adaptive-sparse-inference"},{"arxiv_id":"2310.03744","page":"/paper/improved-baselines-with-visual-instruction"},{"arxiv_id":"2309.09958","page":"/paper/an-empirical-study-of-scaling-instruct-tuned"},{"arxiv_id":"2407.15841","page":"/paper/slowfast-llava-a-strong-training-free"},{"arxiv_id":"2412.01818","page":"/paper/cls-attention-is-all-you-need-for-training"},{"arxiv_id":"2503.15621","page":"/paper/llava-more-a-comparative-study-of-llms-and"},{"arxiv_id":"2410.22313","page":"/paper/senna-bridging-large-vision-language-models"},{"arxiv_id":"2412.02158","page":"/paper/agri-llava-knowledge-infused-large-multimodal"},{"arxiv_id":"2410.02080","page":"/paper/emma-efficient-visual-alignment-in-multi"},{"arxiv_id":"2602.23699","page":"/paper/arxiv-2602-23699"}],"arg_sig_recorded":[["tensor",3,"float","float32"]],"scalar_args":{"original_size":"(8, 6)"},"class_bearing":false},{"code_sha256_prefix":"f5183d69498ae311","path":"transformers/src/transformers/models/llava_next/modeling_llava_next.py","papers":["2408.10945"],"paper_pages":[{"arxiv_id":"2408.10945","page":"/paper/hired-attention-guided-token-dropping-for"}],"arg_sig_recorded":[["tensor",3,"float","float32"]],"scalar_args":{"original_size":"(4, 6)"},"class_bearing":false}]}]}]}