{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-video","entry":"load_video","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":20,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":17,"n_samples_ran":5,"n_samples_fingerprinted":0,"n_places":20,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":1,"ran":4,"unverified":12},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2603.25716","paper":"/paper/arxiv-2603-25716","title":"Out of Sight but Not Out of Mind: Hybrid Memory for Dynamic Video World Models","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"H-EmbodVis/HyDRA","path":"eval_dsc.py","file_url":"https://github.com/H-EmbodVis/HyDRA/blob/HEAD/eval_dsc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a618aac4fab6597f","mcp_get_code":{"code_sha256":"a618aac4fab6597f"}},{"arxiv_id":"2506.09344","paper":"/paper/ming-omni-a-unified-multimodal-model-for","title":"Ming-Omni: A Unified Multimodal Model for Perception and Generation","date":"2025-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"inclusionai/ming","path":"bailingmm_utils_video.py","file_url":"https://github.com/inclusionai/ming/blob/HEAD/bailingmm_utils_video.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3d3e22e2facd5ccf","mcp_get_code":{"code_sha256":"3d3e22e2facd5ccf"}},{"arxiv_id":"2506.00329","paper":null,"title":"arXiv:2506.00329","date":null,"month_inferred_from_arxiv_id":"2025-06","title_source":null,"repo":"STAR-Laboratory/foresight","path":"eval/foresight/common_metrics/batch_eval.py","file_url":"https://github.com/STAR-Laboratory/foresight/blob/HEAD/eval/foresight/common_metrics/batch_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3dca7220b4552a32","mcp_get_code":{"code_sha256":"3dca7220b4552a32"}},{"arxiv_id":"2504.14899","paper":"/paper/uni3c-unifying-precisely-3d-enhanced-camera","title":"Uni3C: Unifying Precisely 3D-Enhanced Camera and Human Motion Controls for Video Generation","date":"2025-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ewrfcas/uni3c","path":"src/utils.py","file_url":"https://github.com/ewrfcas/uni3c/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"522fd30040cae1fc","mcp_get_code":{"code_sha256":"522fd30040cae1fc"}},{"arxiv_id":"2503.19903","paper":"/paper/scaling-vision-pre-training-to-4k-resolution","title":"Scaling Vision Pre-Training to 4K Resolution","date":"2025-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"efficient-large-model/vila","path":"server.py","file_url":"https://github.com/efficient-large-model/vila/blob/HEAD/server.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fca9a6d2afa23288","mcp_get_code":{"code_sha256":"fca9a6d2afa23288"}},{"arxiv_id":"2411.17991","paper":"/paper/videollm-knows-when-to-speak-enhancing-time","title":"VideoLLM Knows When to Speak: Enhancing Time-Sensitive Video Comprehension with Video-Text Duet Interaction Format","date":"2024-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yellow-binary-tree/mmduet","path":"demo/liveinfer.py","file_url":"https://github.com/yellow-binary-tree/mmduet/blob/HEAD/demo/liveinfer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3970e26740ae47a6","mcp_get_code":{"code_sha256":"3970e26740ae47a6"}},{"arxiv_id":"2411.11409","paper":"/paper/ikea-manuals-at-work-4d-grounding-of-assembly","title":"IKEA Manuals at Work: 4D Grounding of Assembly Instructions on Internet Videos","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yunongLiu1/IKEA-Manuals-at-Work","path":"src/IKEAVideo/dataloader/assembly_video.py","file_url":"https://github.com/yunongLiu1/IKEA-Manuals-at-Work/blob/HEAD/src/IKEAVideo/dataloader/assembly_video.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4c6cff895aa90b9b","mcp_get_code":{"code_sha256":"4c6cff895aa90b9b"}},{"arxiv_id":"2411.01505","paper":"/paper/object-segmentation-from-common-fate-motion","title":"Object segmentation from common fate: Motion energy processing enables human-like zero-shot generalization to random dot stimuli","date":"2024-11-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mtangemann/motion_energy_segmentation","path":"motion_energy_segmentation/io_utils.py","file_url":"https://github.com/mtangemann/motion_energy_segmentation/blob/HEAD/motion_energy_segmentation/io_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a964f0bf9f047ba8","mcp_get_code":{"code_sha256":"a964f0bf9f047ba8"}},{"arxiv_id":"2409.12319","paper":"/paper/large-language-models-are-strong-audio-visual","title":"Large Language Models are Strong Audio-Visual Speech Recognition Learners","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umbertocappellazzo/llama-avsr","path":"datamodule/av_dataset.py","file_url":"https://github.com/umbertocappellazzo/llama-avsr/blob/HEAD/datamodule/av_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e3ffd0b6ffbdf70f","mcp_get_code":{"code_sha256":"e3ffd0b6ffbdf70f"}},{"arxiv_id":"2407.03563","paper":"/paper/learning-video-temporal-dynamics-with-cross","title":"Learning Video Temporal Dynamics with Cross-Modal Attention for Robust Audio-Visual Speech Recognition","date":"2024-07-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sungnyun/avsr-temporal-dynamics","path":"avhubert/utils.py","file_url":"https://github.com/sungnyun/avsr-temporal-dynamics/blob/HEAD/avhubert/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d5a7afb95212031e","mcp_get_code":{"code_sha256":"d5a7afb95212031e"}},{"arxiv_id":"2405.19298","paper":"/paper/adaptive-image-quality-assessment-via","title":"Adaptive Image Quality Assessment via Teaching Large Multimodal Model to Compare","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Q-Future/Compare2Score","path":"q_align/load_video.py","file_url":"https://github.com/Q-Future/Compare2Score/blob/HEAD/q_align/load_video.py","status":"ran_fixture","verification_level":1,"contract_check":"DEP_MISSING","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"72950fb8c1bfc7a8","mcp_get_code":{"code_sha256":"72950fb8c1bfc7a8"}},{"arxiv_id":"2402.15151","paper":"/paper/where-visual-speech-meets-language-vsp-llm","title":"Where Visual Speech Meets Language: VSP-LLM Framework for Efficient and Context-Aware Visual Speech Processing","date":"2024-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sally-sh/vsp-llm","path":"src/utils_vsp_llm.py","file_url":"https://github.com/sally-sh/vsp-llm/blob/HEAD/src/utils_vsp_llm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d5a7afb95212031e","mcp_get_code":{"code_sha256":"d5a7afb95212031e"}},{"arxiv_id":"2401.10226","paper":"/paper/towards-language-driven-video-inpainting-via","title":"Towards Language-Driven Video Inpainting via Multimodal Large Language Models","date":"2024-01-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jianzongwu/Language-Driven-Video-Inpainting","path":"inference_interactive.py","file_url":"https://github.com/jianzongwu/Language-Driven-Video-Inpainting/blob/HEAD/inference_interactive.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a612419f9fdb01ba","mcp_get_code":{"code_sha256":"a612419f9fdb01ba"}},{"arxiv_id":"2401.01651","paper":"/paper/aigcbench-comprehensive-evaluation-of-image","title":"AIGCBench: Comprehensive Evaluation of Image-to-Video Content Generated by AI","date":"2024-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"benchcouncil/aigcbench","path":"utils.py","file_url":"https://github.com/benchcouncil/aigcbench/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"568153c42a899735","mcp_get_code":{"code_sha256":"568153c42a899735"}},{"arxiv_id":"2312.17090","paper":"/paper/q-align-teaching-lmms-for-visual-scoring-via","title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","date":"2023-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"q-future/q-align","path":"q_align/evaluate/scorer.py","file_url":"https://github.com/q-future/q-align/blob/HEAD/q_align/evaluate/scorer.py","status":"ran_fixture","verification_level":1,"contract_check":"DEP_MISSING","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"72950fb8c1bfc7a8","mcp_get_code":{"code_sha256":"72950fb8c1bfc7a8"}},{"arxiv_id":"2312.07531","paper":"/paper/wham-reconstructing-world-grounded-humans","title":"WHAM: Reconstructing World-grounded Humans with Accurate 3D Motion","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yohanshin/WHAM","path":"wham_api.py","file_url":"https://github.com/yohanshin/WHAM/blob/HEAD/wham_api.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f5e1eddc0f7791f2","mcp_get_code":{"code_sha256":"f5e1eddc0f7791f2"}},{"arxiv_id":"2310.16640","paper":"/paper/emoclip-a-vision-language-method-for-zero","title":"EmoCLIP: A Vision-Language Method for Zero-Shot Video Facial Expression Recognition","date":"2023-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nickyfot/emoclip","path":"DataLoaders/utils.py","file_url":"https://github.com/nickyfot/emoclip/blob/HEAD/DataLoaders/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b7d567db51876aea","mcp_get_code":{"code_sha256":"b7d567db51876aea"}},{"arxiv_id":"2308.09592","paper":"/paper/stablevideo-text-driven-consistency-aware","title":"StableVideo: Text-driven Consistency-aware Diffusion Video Editing","date":"2023-08-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rese1f/stablevideo","path":"stablevideo/atlas_utils.py","file_url":"https://github.com/rese1f/stablevideo/blob/HEAD/stablevideo/atlas_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b73c0affb20bb803","mcp_get_code":{"code_sha256":"b73c0affb20bb803"}},{"arxiv_id":"2303.14307","paper":"/paper/auto-avsr-audio-visual-speech-recognition","title":"Auto-AVSR: Audio-Visual Speech Recognition with Automatic Labels","date":"2023-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mpc001/auto_avsr","path":"datamodule/av_dataset.py","file_url":"https://github.com/mpc001/auto_avsr/blob/HEAD/datamodule/av_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e3ffd0b6ffbdf70f","mcp_get_code":{"code_sha256":"e3ffd0b6ffbdf70f"}},{"arxiv_id":"1611.09326","paper":"/paper/the-one-hundred-layers-tiramisu-fully","title":"The One Hundred Layers Tiramisu: Fully Convolutional DenseNets for Semantic Segmentation","date":"2016-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ankit-vaghela30/Cilia-Segmentation","path":"hastings/io_support.py","file_url":"https://github.com/ankit-vaghela30/Cilia-Segmentation/blob/HEAD/hastings/io_support.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6b7485606e8ec4b4","mcp_get_code":{"code_sha256":"6b7485606e8ec4b4"}}]}