{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/downsample-basic-block","entry":"downsample_basic_block","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":26,"n_papers_ran":25,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":6,"n_samples_ran":5,"n_samples_fingerprinted":1,"n_places":26,"n_places_pointer_only":10,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":3,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2409.12319","paper":"/paper/large-language-models-are-strong-audio-visual","title":"Large Language Models are Strong Audio-Visual Speech Recognition Learners","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umbertocappellazzo/llama-avsr","path":"av_hubert/avhubert/resnet.py","file_url":"https://github.com/umbertocappellazzo/llama-avsr/blob/HEAD/av_hubert/avhubert/resnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d50e3a8f0b0f9a21","mcp_get_code":{"code_sha256":"d50e3a8f0b0f9a21"}},{"arxiv_id":"2407.19308","paper":"/paper/comprehensive-attribution-inherently","title":"Comprehensive Attribution: Inherently Explainable Vision Model with Feature Detector","date":"2024-07-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zood123/comet","path":"models/ResNetWithGate.py","file_url":"https://github.com/zood123/comet/blob/HEAD/models/ResNetWithGate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b51274b9d6c9c97a","mcp_get_code":{"code_sha256":"b51274b9d6c9c97a"}},{"arxiv_id":"2407.03563","paper":"/paper/learning-video-temporal-dynamics-with-cross","title":"Learning Video Temporal Dynamics with Cross-Modal Attention for Robust Audio-Visual Speech Recognition","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sungnyun/avsr-temporal-dynamics","path":"avhubert/resnet.py","file_url":"https://github.com/sungnyun/avsr-temporal-dynamics/blob/HEAD/avhubert/resnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d50e3a8f0b0f9a21","mcp_get_code":{"code_sha256":"d50e3a8f0b0f9a21"}},{"arxiv_id":"2406.10082","paper":"/paper/whisper-flamingo-integrating-visual-features","title":"Whisper-Flamingo: Integrating Visual Features into Whisper for Audio-Visual Speech Recognition and Translation","date":"2024-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"roudimit/whisper-flamingo","path":"whisper/resnet.py","file_url":"https://github.com/roudimit/whisper-flamingo/blob/HEAD/whisper/resnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d50e3a8f0b0f9a21","mcp_get_code":{"code_sha256":"d50e3a8f0b0f9a21"}},{"arxiv_id":"2405.09539","paper":"/paper/mmfusion-multi-modality-diffusion-model-for","title":"MMFusion: Multi-modality Diffusion Model for Lymph Node Metastasis Diagnosis in Esophageal Cancer","date":"2024-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wuchengyu123/mmfusion","path":"model/resnet.py","file_url":"https://github.com/wuchengyu123/mmfusion/blob/HEAD/model/resnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"03e03bbdf5410afc","mcp_get_code":{"code_sha256":"03e03bbdf5410afc"}},{"arxiv_id":"2402.15151","paper":"/paper/where-visual-speech-meets-language-vsp-llm","title":"Where Visual Speech Meets Language: VSP-LLM Framework for Efficient and Context-Aware Visual Speech Processing","date":"2024-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sally-sh/vsp-llm","path":"src/resnet.py","file_url":"https://github.com/sally-sh/vsp-llm/blob/HEAD/src/resnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d50e3a8f0b0f9a21","mcp_get_code":{"code_sha256":"d50e3a8f0b0f9a21"}},{"arxiv_id":"2401.03468","paper":"/paper/multichannel-av-wav2vec2-a-framework-for","title":"Multichannel AV-wav2vec2: A Framework for Learning Multichannel Multi-Modal Speech Representation","date":"2024-01-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zqs01/multi-channel-wav2vec2","path":"avhubert/resnet.py","file_url":"https://github.com/zqs01/multi-channel-wav2vec2/blob/HEAD/avhubert/resnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d50e3a8f0b0f9a21","mcp_get_code":{"code_sha256":"d50e3a8f0b0f9a21"}},{"arxiv_id":"2312.02512","paper":"/paper/av2av-direct-audio-visual-speech-to-audio","title":"AV2AV: Direct Audio-Visual Speech to Audio-Visual Speech Translation with Unified Audio-Visual Speech Representation","date":"2023-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"choijeongsoo/av2av","path":"av2unit/avhubert/resnet.py","file_url":"https://github.com/choijeongsoo/av2av/blob/HEAD/av2unit/avhubert/resnet.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d50e3a8f0b0f9a21","mcp_get_code":{"code_sha256":"d50e3a8f0b0f9a21"}},{"arxiv_id":"2311.12052","paper":"/paper/magicdance-realistic-human-dance-video","title":"MagicPose: Realistic Human Poses and Facial Expressions Retargeting with Identity-aware Diffusion","date":"2023-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Boese0601/MagicDance","path":"tool/metrics/resnet3d.py","file_url":"https://github.com/Boese0601/MagicDance/blob/HEAD/tool/metrics/resnet3d.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2307.16184","paper":"/paper/unified-model-for-image-video-audio-and","title":"UnIVAL: Unified Model for Image, Video, Audio and Language Tasks","date":"2023-07-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mshukor/unival","path":"models/unival/encoders/resnext3d.py","file_url":"https://github.com/mshukor/unival/blob/HEAD/models/unival/encoders/resnext3d.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2307.14126","paper":"/paper/multi-modal-learning-with-missing-modality-1","title":"Multi-modal Learning with Missing Modality via Shared-Specific Feature Modelling","date":"2023-07-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"billhhh/ShaSpec","path":"DualNet_SS.py","file_url":"https://github.com/billhhh/ShaSpec/blob/HEAD/DualNet_SS.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"285c8c634525802b","mcp_get_code":{"code_sha256":"285c8c634525802b"}},{"arxiv_id":"2211.04031","paper":"/paper/hilbert-distillation-for-cross-dimensionality","title":"Hilbert Distillation for Cross-Dimensionality Networks","date":"2022-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"EagleMIT/Hilbert-Distillation","path":"model/resnet.py","file_url":"https://github.com/EagleMIT/Hilbert-Distillation/blob/HEAD/model/resnet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2211.03779","paper":"/paper/cripp-vqa-counterfactual-reasoning-about","title":"CRIPP-VQA: Counterfactual Reasoning about Implicit Physical Properties via Video Question Answering","date":"2022-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thaolmk54/hcrn-videoqa","path":"preprocess/models/pre_act_resnet.py","file_url":"https://github.com/thaolmk54/hcrn-videoqa/blob/HEAD/preprocess/models/pre_act_resnet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2105.05226","paper":"/paper/home-action-genome-cooperative-compositional","title":"Home Action Genome: Cooperative Compositional Action Understanding","date":"2021-05-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nishantrai18/homage","path":"backbone/resnet_2d3d.py","file_url":"https://github.com/nishantrai18/homage/blob/HEAD/backbone/resnet_2d3d.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2008.11516","paper":"/paper/making-a-case-for-3d-convolutions-for-object","title":"Making a Case for 3D Convolutions for Object Segmentation in Videos","date":"2020-08-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sabarim/3DC-Seg","path":"network/Resnet3d.py","file_url":"https://github.com/sabarim/3DC-Seg/blob/HEAD/network/Resnet3d.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2008.01065","paper":"/paper/memory-augmented-dense-predictive-coding-for","title":"Memory-augmented Dense Predictive Coding for Video Representation Learning","date":"2020-08-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TengdaHan/MemDPC","path":"backbone/resnet_2d3d.py","file_url":"https://github.com/TengdaHan/MemDPC/blob/HEAD/backbone/resnet_2d3d.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2007.12530","paper":"/paper/a-comprehensive-study-on-sign-language","title":"A Comprehensive Study on Deep Learning-based Methods for Sign Language Recognition","date":"2020-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iliasprc/slrzoo","path":"models/resnet3d.py","file_url":"https://github.com/iliasprc/slrzoo/blob/HEAD/models/resnet3d.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2004.11362","paper":"/paper/supervised-contrastive-learning","title":"Supervised Contrastive Learning","date":"2020-04-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"davidczy/supcon_gamma","path":"networks/ThreeDResnet.py","file_url":"https://github.com/davidczy/supcon_gamma/blob/HEAD/networks/ThreeDResnet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"2001.06499","paper":"/paper/temporal-interlacing-network","title":"Temporal Interlacing Network","date":"2020-01-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eynaij/X-Temporal_catdim","path":"x_temporal/models/resnet3D.py","file_url":"https://github.com/eynaij/X-Temporal_catdim/blob/HEAD/x_temporal/models/resnet3D.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"1912.09930","paper":"/paper/something-else-compositional-action","title":"Something-Else: Compositional Action Recognition with Spatial-Temporal Interaction Networks","date":"2019-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joaanna/something_else","path":"code/model/resnet3d_xl.py","file_url":"https://github.com/joaanna/something_else/blob/HEAD/code/model/resnet3d_xl.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"1912.05534","paper":"/paper/why-cant-i-dance-in-the-mall-learning-to-1","title":"Why Can't I Dance in the Mall? Learning to Mitigate Scene Bias in Action Recognition","date":"2019-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vt-vl-lab/SDN","path":"models/pre_act_resnet.py","file_url":"https://github.com/vt-vl-lab/SDN/blob/HEAD/models/pre_act_resnet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"1911.00232","paper":"/paper/multi-moments-in-time-learning-and","title":"Multi-Moments in Time: Learning and Interpreting Models for Multi-Action Video Understanding","date":"2019-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"metalbubble/moments_models","path":"models.py","file_url":"https://github.com/metalbubble/moments_models/blob/HEAD/models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":false,"code_sha256_prefix":"8ad16556fa2664bf","mcp_get_code":{"code_sha256":"8ad16556fa2664bf"}},{"arxiv_id":"1909.04656","paper":"/paper/video-representation-learning-by-dense","title":"Video Representation Learning by Dense Predictive Coding","date":"2019-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"TengdaHan/DPC","path":"backbone/resnet_2d3d.py","file_url":"https://github.com/TengdaHan/DPC/blob/HEAD/backbone/resnet_2d3d.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"1908.01373","paper":"/paper/unsupervised-microvascular-image-segmentation","title":"Unsupervised Microvascular Image Segmentation Using an Active Contours Mimicking Neural Network","date":"2019-08-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shirgur/UMIS","path":"networks/resnet.py","file_url":"https://github.com/shirgur/UMIS/blob/HEAD/networks/resnet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"1711.09577","paper":"/paper/can-spatiotemporal-3d-cnns-retrace-the","title":"Can Spatiotemporal 3D CNNs Retrace the History of 2D CNNs and ImageNet?","date":"2017-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianhai123/3D-ResNets","path":"models/resnet.py","file_url":"https://github.com/tianhai123/3D-ResNets/blob/HEAD/models/resnet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}},{"arxiv_id":"1705.07750","paper":"/paper/quo-vadis-action-recognition-a-new-model-and","title":"Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset","date":"2017-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"43c0211ff8dbca5f","mcp_get_code":{"code_sha256":"43c0211ff8dbca5f"}}]}