{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-generation/papers/3","list_of":"/task/video-generation","task":"Video Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":15,"rows_per_page":100,"rows":[201,300],"of":1466,"counts":{"archive_papers_tagged":1466,"with_a_code_link":609,"where_syntology_ran_a_sample":257,"not_listed_spam_title":0,"listed":1466,"listed_where_code_ran":257,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":221,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":221,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-generation","prev":"/task/video-generation/papers/2","next":"/task/video-generation/papers/4","papers":[{"url":"/paper/magic-1-for-1-generating-one-minute-video","slug":"magic-1-for-1-generating-one-minute-video","title":"Magic 1-For-1: Generating One Minute Video Clips within One Minute","date":"2025-02-11","arxiv_id":"2502.07701","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-diffusion-model-with-spatial","slug":"conditional-diffusion-model-with-spatial","title":"Conditional diffusion model with spatial attention and latent embedding for medical image segmentation","date":"2025-02-10","arxiv_id":"2502.06997","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-vdit-efficient-video-diffusion","slug":"efficient-vdit-efficient-video-diffusion","title":"Efficient-vDiT: Efficient Video Diffusion Transformers With Attention Tile","date":"2025-02-10","arxiv_id":"2502.06155","repositories_listed":1,"syntology":null},{"url":"/paper/history-guided-video-diffusion","slug":"history-guided-video-diffusion","title":"History-Guided Video Diffusion","date":"2025-02-10","arxiv_id":"2502.06764","repositories_listed":1,"syntology":null},{"url":"/paper/a-physical-coherence-benchmark-for-evaluating","slug":"a-physical-coherence-benchmark-for-evaluating","title":"A Physical Coherence Benchmark for Evaluating Video Generation Models via Optical Flow-guided Frame Prediction","date":"2025-02-08","arxiv_id":"2502.05503","repositories_listed":1,"syntology":null},{"url":"/paper/flashvideo-flowing-fidelity-to-detail-for","slug":"flashvideo-flowing-fidelity-to-detail-for","title":"FlashVideo:Flowing Fidelity to Detail for Efficient High-Resolution Video Generation","date":"2025-02-07","arxiv_id":"2502.05179","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":3,"n_no_contract":8,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 3 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/flashvideo-flowing-fidelity-to-detail-for#ran","syntology_url":"https://syntology.ai/paper/2502.05179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05179"}},"official":{"repos":["foundationvision/flashvideo"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/goku-flow-based-video-generative-foundation","slug":"goku-flow-based-video-generative-foundation","title":"Goku: Flow Based Video Generative Foundation Models","date":"2025-02-07","arxiv_id":"2502.04896","repositories_listed":1,"syntology":null},{"url":"/paper/content-rich-aigc-video-quality-assessment","slug":"content-rich-aigc-video-quality-assessment","title":"Content-Rich AIGC Video Quality Assessment via Intricate Text Alignment and Motion-Aware Consistency","date":"2025-02-06","arxiv_id":"2502.04076","repositories_listed":1,"syntology":null},{"url":"/paper/fast-video-generation-with-sliding-tile","slug":"fast-video-generation-with-sliding-tile","title":"Fast Video Generation with Sliding Tile Attention","date":"2025-02-06","arxiv_id":"2502.04507","repositories_listed":1,"syntology":null},{"url":"/paper/on-device-sora-enabling-diffusion-based-text","slug":"on-device-sora-enabling-diffusion-based-text","title":"On-device Sora: Enabling Training-Free Diffusion-based Text-to-Video Generation for Mobile Devices","date":"2025-02-05","arxiv_id":"2502.04363","repositories_listed":1,"syntology":null},{"url":"/paper/improved-training-technique-for-latent","slug":"improved-training-technique-for-latent","title":"Improved Training Technique for Latent Consistency Models","date":"2025-02-03","arxiv_id":"2502.01441","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improved-training-technique-for-latent#ran","syntology_url":"https://syntology.ai/paper/2502.01441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01441"}},"official":{"repos":["quandao10/slct"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vidsketch-hand-drawn-sketch-driven-video","slug":"vidsketch-hand-drawn-sketch-driven-video","title":"VidSketch: Hand-drawn Sketch-Driven Video Generation with Diffusion Control","date":"2025-02-03","arxiv_id":"2502.01101","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vidsketch-hand-drawn-sketch-driven-video#ran","syntology_url":"https://syntology.ai/paper/2502.01101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.01101"}},"official":{"repos":["CSfufu/VidSketch"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/vilp-imitation-learning-with-latent-video","slug":"vilp-imitation-learning-with-latent-video","title":"VILP: Imitation Learning with Latent Video Planning","date":"2025-02-03","arxiv_id":"2502.01784","repositories_listed":1,"syntology":null},{"url":"/paper/inference-time-text-to-video-alignment-with","slug":"inference-time-text-to-video-alignment-with","title":"Inference-Time Text-to-Video Alignment with Diffusion Latent Beam Search","date":"2025-01-31","arxiv_id":"2501.19252","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 3 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inference-time-text-to-video-alignment-with#ran","syntology_url":"https://syntology.ai/paper/2501.19252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19252"}},"official":null}},{"url":"/paper/cascadev-an-implementation-of-wurstchen","slug":"cascadev-an-implementation-of-wurstchen","title":"CascadeV: An Implementation of Wurstchen Architecture for Video Generation","date":"2025-01-28","arxiv_id":"2501.16612","repositories_listed":1,"syntology":null},{"url":"/paper/videoshield-regulating-diffusion-based-video","slug":"videoshield-regulating-diffusion-based-video","title":"VideoShield: Regulating Diffusion-based Video Generation Models via Watermarking","date":"2025-01-24","arxiv_id":"2501.14195","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videoshield-regulating-diffusion-based-video#ran","syntology_url":"https://syntology.ai/paper/2501.14195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14195"}},"official":{"repos":["hurunyi/videoshield"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/echovideo-identity-preserving-human-video","slug":"echovideo-identity-preserving-human-video","title":"EchoVideo: Identity-Preserving Human Video Generation by Multimodal Feature Fusion","date":"2025-01-23","arxiv_id":"2501.13452","repositories_listed":1,"syntology":null},{"url":"/paper/video-depth-anything-consistent-depth","slug":"video-depth-anything-consistent-depth","title":"Video Depth Anything: Consistent Depth Estimation for Super-Long Videos","date":"2025-01-21","arxiv_id":"2501.12375","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-depth-anything-consistent-depth#ran","syntology_url":"https://syntology.ai/paper/2501.12375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.12375"}},"official":null}},{"url":"/paper/catv2ton-taming-diffusion-transformers-for","slug":"catv2ton-taming-diffusion-transformers-for","title":"CatV2TON: Taming Diffusion Transformers for Vision-Based Virtual Try-On with Temporal Concatenation","date":"2025-01-20","arxiv_id":"2501.11325","repositories_listed":1,"syntology":null},{"url":"/paper/diffueraser-a-diffusion-model-for-video","slug":"diffueraser-a-diffusion-model-for-video","title":"DiffuEraser: A Diffusion Model for Video Inpainting","date":"2025-01-17","arxiv_id":"2501.10018","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffueraser-a-diffusion-model-for-video#ran","syntology_url":"https://syntology.ai/paper/2501.10018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.10018"}},"official":{"repos":["lixiaowen-xw/diffueraser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-generative-video-models-learn-physical","slug":"do-generative-video-models-learn-physical","title":"Do generative video models understand physical principles?","date":"2025-01-14","arxiv_id":"2501.09038","repositories_listed":1,"syntology":null},{"url":"/paper/framepainter-endowing-interactive-image","slug":"framepainter-endowing-interactive-image","title":"FramePainter: Endowing Interactive Image Editing with Video Diffusion Priors","date":"2025-01-14","arxiv_id":"2501.08225","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/framepainter-endowing-interactive-image#ran","syntology_url":"https://syntology.ai/paper/2501.08225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08225"}},"official":{"repos":["ybybzhang/framepainter"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vchitect-2-0-parallel-transformer-for-scaling","slug":"vchitect-2-0-parallel-transformer-for-scaling","title":"Vchitect-2.0: Parallel Transformer for Scaling Up Video Diffusion Models","date":"2025-01-14","arxiv_id":"2501.08453","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vchitect-2-0-parallel-transformer-for-scaling#ran","syntology_url":"https://syntology.ai/paper/2501.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08453"}},"official":null}},{"url":"/paper/diffusion-as-shader-3d-aware-video-diffusion","slug":"diffusion-as-shader-3d-aware-video-diffusion","title":"Diffusion as Shader: 3D-aware Video Diffusion for Versatile Video Generation Control","date":"2025-01-07","arxiv_id":"2501.03847","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-as-shader-3d-aware-video-diffusion#ran","syntology_url":"https://syntology.ai/paper/2501.03847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03847"}},"official":{"repos":["igl-hkust/diffusionasshader"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/magic-mirror-id-preserved-video-generation-in","slug":"magic-mirror-id-preserved-video-generation-in","title":"Magic Mirror: ID-Preserved Video Generation in Video Diffusion Transformers","date":"2025-01-07","arxiv_id":"2501.03931","repositories_listed":1,"syntology":null},{"url":"/paper/transpixar-advancing-text-to-video-generation","slug":"transpixar-advancing-text-to-video-generation","title":"TransPixeler: Advancing Text-to-Video Generation with Transparency","date":"2025-01-06","arxiv_id":"2501.03006","repositories_listed":1,"syntology":null},{"url":"/paper/joygen-audio-driven-3d-depth-aware-talking","slug":"joygen-audio-driven-3d-depth-aware-talking","title":"JoyGen: Audio-Driven 3D Depth-Aware Talking-Face Video Editing","date":"2025-01-03","arxiv_id":"2501.01798","repositories_listed":1,"syntology":null},{"url":"/paper/ltx-video-realtime-video-latent-diffusion","slug":"ltx-video-realtime-video-latent-diffusion","title":"LTX-Video: Realtime Video Latent Diffusion","date":"2024-12-30","arxiv_id":"2501.00103","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ltx-video-realtime-video-latent-diffusion#ran","syntology_url":"https://syntology.ai/paper/2501.00103","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00103"}},"official":{"repos":["Lightricks/LTX-Video"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vinci-a-real-time-embodied-smart-assistant","slug":"vinci-a-real-time-embodied-smart-assistant","title":"Vinci: A Real-time Embodied Smart Assistant based on Egocentric Vision-Language Model","date":"2024-12-30","arxiv_id":"2412.21080","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vinci-a-real-time-embodied-smart-assistant#ran","syntology_url":"https://syntology.ai/paper/2412.21080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.21080"}},"official":{"repos":["opengvlab/vinci"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/visionreward-fine-grained-multi-dimensional","slug":"visionreward-fine-grained-multi-dimensional","title":"VisionReward: Fine-Grained Multi-Dimensional Human Preference Learning for Image and Video Generation","date":"2024-12-30","arxiv_id":"2412.21059","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visionreward-fine-grained-multi-dimensional#ran","syntology_url":"https://syntology.ai/paper/2412.21059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.21059"}},"official":{"repos":["thudm/visionreward"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/drivingworld-constructingworld-model-for","slug":"drivingworld-constructingworld-model-for","title":"DrivingWorld: Constructing World Model for Autonomous Driving via Video GPT","date":"2024-12-27","arxiv_id":"2412.19505","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drivingworld-constructingworld-model-for#ran","syntology_url":"https://syntology.ai/paper/2412.19505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.19505"}},"official":{"repos":["yvanyin/drivingworld"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videomaker-zero-shot-customized-video","slug":"videomaker-zero-shot-customized-video","title":"VideoMaker: Zero-shot Customized Video Generation with the Inherent Force of Video Diffusion Models","date":"2024-12-27","arxiv_id":"2412.19645","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomaker-zero-shot-customized-video#ran","syntology_url":"https://syntology.ai/paper/2412.19645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.19645"}},"official":null}},{"url":"/paper/ditctrl-exploring-attention-control-in-multi","slug":"ditctrl-exploring-attention-control-in-multi","title":"DiTCtrl: Exploring Attention Control in Multi-Modal Diffusion Transformer for Tuning-Free Multi-Prompt Longer Video Generation","date":"2024-12-24","arxiv_id":"2412.18597","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ditctrl-exploring-attention-control-in-multi#ran","syntology_url":"https://syntology.ai/paper/2412.18597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18597"}},"official":{"repos":["tencentarc/ditctrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vidtwin-video-vae-with-decoupled-structure","slug":"vidtwin-video-vae-with-decoupled-structure","title":"VidTwin: Video VAE with Decoupled Structure and Dynamics","date":"2024-12-23","arxiv_id":"2412.17726","repositories_listed":1,"syntology":null},{"url":"/paper/customttt-motion-and-appearance-customized","slug":"customttt-motion-and-appearance-customized","title":"CustomTTT: Motion and Appearance Customized Video Generation via Test-Time Training","date":"2024-12-20","arxiv_id":"2412.15646","repositories_listed":1,"syntology":null},{"url":"/paper/consistent-human-image-and-video-generation","slug":"consistent-human-image-and-video-generation","title":"Consistent Human Image and Video Generation with Spatially Conditioned Diffusion","date":"2024-12-19","arxiv_id":"2412.14531","repositories_listed":1,"syntology":null},{"url":"/paper/prompt-a-video-prompt-your-video-diffusion","slug":"prompt-a-video-prompt-your-video-diffusion","title":"Prompt-A-Video: Prompt Your Video Diffusion Model via Preference-Aligned LLM","date":"2024-12-19","arxiv_id":"2412.15156","repositories_listed":1,"syntology":null},{"url":"/paper/video-prediction-policy-a-generalist-robot","slug":"video-prediction-policy-a-generalist-robot","title":"Video Prediction Policy: A Generalist Robot Policy with Predictive Visual Representations","date":"2024-12-19","arxiv_id":"2412.14803","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/video-prediction-policy-a-generalist-robot#ran","syntology_url":"https://syntology.ai/paper/2412.14803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14803"}},"official":null}},{"url":"/paper/autoregressive-video-generation-without","slug":"autoregressive-video-generation-without","title":"Autoregressive Video Generation without Vector Quantization","date":"2024-12-18","arxiv_id":"2412.14169","repositories_listed":1,"syntology":{"n":26,"n_ran":19,"n_constructed":18,"n_ran_checked":19,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":0,"phrase":"19 ran (of which 18 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/autoregressive-video-generation-without#ran","syntology_url":"https://syntology.ai/paper/2412.14169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14169"}},"official":{"repos":["baaivision/nova"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":18,"n_ran_no_instrument_failure":19,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/real-time-one-step-diffusion-based-expressive","slug":"real-time-one-step-diffusion-based-expressive","title":"Real-time One-Step Diffusion-based Expressive Portrait Videos Generation","date":"2024-12-18","arxiv_id":"2412.13479","repositories_listed":1,"syntology":null},{"url":"/paper/vidtok-a-versatile-and-open-source-video","slug":"vidtok-a-versatile-and-open-source-video","title":"VidTok: A Versatile and Open-Source Video Tokenizer","date":"2024-12-17","arxiv_id":"2412.13061","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vidtok-a-versatile-and-open-source-video#ran","syntology_url":"https://syntology.ai/paper/2412.13061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13061"}},"official":{"repos":["microsoft/vidtok"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-inbetweening-through-frame-wise","slug":"generative-inbetweening-through-frame-wise","title":"Generative Inbetweening through Frame-wise Conditions-Driven Video Generation","date":"2024-12-16","arxiv_id":"2412.11755","repositories_listed":1,"syntology":null},{"url":"/paper/vg-tvp-multimodal-procedural-planning-via","slug":"vg-tvp-multimodal-procedural-planning-via","title":"VG-TVP: Multimodal Procedural Planning via Visually Grounded Text-Video Prompting","date":"2024-12-16","arxiv_id":"2412.11621","repositories_listed":1,"syntology":null},{"url":"/paper/video-diffusion-transformers-are-in-context","slug":"video-diffusion-transformers-are-in-context","title":"Video Diffusion Transformers are In-Context Learners","date":"2024-12-14","arxiv_id":"2412.10783","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-frontiers-of-animation-video","slug":"exploring-the-frontiers-of-animation-video","title":"AniSora: Exploring the Frontiers of Animation Video Generation in the Sora Era","date":"2024-12-13","arxiv_id":"2412.10255","repositories_listed":1,"syntology":null},{"url":"/paper/doe-1-closed-loop-autonomous-driving-with","slug":"doe-1-closed-loop-autonomous-driving-with","title":"Doe-1: Closed-Loop Autonomous Driving with Large World Model","date":"2024-12-12","arxiv_id":"2412.09627","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doe-1-closed-loop-autonomous-driving-with#ran","syntology_url":"https://syntology.ai/paper/2412.09627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09627"}},"official":{"repos":["wzzheng/doe"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/owl-1-omni-world-model-for-consistent-long","slug":"owl-1-omni-world-model-for-consistent-long","title":"Owl-1: Omni World Model for Consistent Long Video Generation","date":"2024-12-12","arxiv_id":"2412.09600","repositories_listed":1,"syntology":null},{"url":"/paper/ufo-enhancing-diffusion-based-video","slug":"ufo-enhancing-diffusion-based-video","title":"UFO: Enhancing Diffusion-Based Video Generation with a Uniform Frame Organizer","date":"2024-12-12","arxiv_id":"2412.09389","repositories_listed":1,"syntology":null},{"url":"/paper/acdit-interpolating-autoregressive","slug":"acdit-interpolating-autoregressive","title":"ACDiT: Interpolating Autoregressive Conditional Modeling and Diffusion Transformer","date":"2024-12-10","arxiv_id":"2412.07720","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/acdit-interpolating-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2412.07720","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07720"}},"official":{"repos":["thunlp/acdit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluation-agent-efficient-and-promptable","slug":"evaluation-agent-efficient-and-promptable","title":"Evaluation Agent: Efficient and Promptable Evaluation Framework for Visual Generative Models","date":"2024-12-10","arxiv_id":"2412.09645","repositories_listed":1,"syntology":null},{"url":"/paper/syncammaster-synchronizing-multi-camera-video","slug":"syncammaster-synchronizing-multi-camera-video","title":"SynCamMaster: Synchronizing Multi-Camera Video Generation from Diverse Viewpoints","date":"2024-12-10","arxiv_id":"2412.07760","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/syncammaster-synchronizing-multi-camera-video#ran","syntology_url":"https://syntology.ai/paper/2412.07760","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07760"}},"official":{"repos":["kwaivgi/syncammaster"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flexdit-dynamic-token-density-control-for","slug":"flexdit-dynamic-token-density-control-for","title":"FlexDiT: Dynamic Token Density Control for Diffusion Transformer","date":"2024-12-08","arxiv_id":"2412.06028","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/flexdit-dynamic-token-density-control-for#ran","syntology_url":"https://syntology.ai/paper/2412.06028","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06028"}},"official":{"repos":["changsn/FlexDiT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/stag-1-towards-realistic-4d-driving","slug":"stag-1-towards-realistic-4d-driving","title":"Stag-1: Towards Realistic 4D Driving Simulation with Video Generation Model","date":"2024-12-06","arxiv_id":"2412.05280","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stag-1-towards-realistic-4d-driving#ran","syntology_url":"https://syntology.ai/paper/2412.05280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05280"}},"official":{"repos":["wzzheng/stag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unimlvg-unified-framework-for-multi-view-long","slug":"unimlvg-unified-framework-for-multi-view-long","title":"UniMLVG: Unified Framework for Multi-view Long Video Generation with Comprehensive Control Capabilities for Autonomous Driving","date":"2024-12-06","arxiv_id":"2412.04842","repositories_listed":1,"syntology":null},{"url":"/paper/divot-diffusion-powers-video-tokenizer-for","slug":"divot-diffusion-powers-video-tokenizer-for","title":"Divot: Diffusion Powers Video Tokenizer for Comprehension and Generation","date":"2024-12-05","arxiv_id":"2412.04432","repositories_listed":1,"syntology":null},{"url":"/paper/navigation-world-models","slug":"navigation-world-models","title":"Navigation World Models","date":"2024-12-04","arxiv_id":"2412.03572","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/navigation-world-models#ran","syntology_url":"https://syntology.ai/paper/2412.03572","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03572"}},"official":null}},{"url":"/paper/videogen-of-thought-a-collaborative-framework","slug":"videogen-of-thought-a-collaborative-framework","title":"VideoGen-of-Thought: A Collaborative Framework for Multi-Shot Video Generation","date":"2024-12-03","arxiv_id":"2412.02259","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videogen-of-thought-a-collaborative-framework#ran","syntology_url":"https://syntology.ai/paper/2412.02259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02259"}},"official":{"repos":["DuNGEOnmassster/VideoGen-of-Thought"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/motion-dreamer-realizing-physically-coherent","slug":"motion-dreamer-realizing-physically-coherent","title":"Motion Dreamer: Realizing Physically Coherent Video Generation through Scene-Aware Motion Reasoning","date":"2024-11-30","arxiv_id":"2412.00547","repositories_listed":1,"syntology":null},{"url":"/paper/phyt2v-llm-guided-iterative-self-refinement","slug":"phyt2v-llm-guided-iterative-self-refinement","title":"PhyT2V: LLM-Guided Iterative Self-Refinement for Physics-Grounded Text-to-Video Generation","date":"2024-11-30","arxiv_id":"2412.00596","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/phyt2v-llm-guided-iterative-self-refinement#ran","syntology_url":"https://syntology.ai/paper/2412.00596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.00596"}},"official":{"repos":["pittisl/phyt2v"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/timestep-embedding-tells-it-s-time-to-cache","slug":"timestep-embedding-tells-it-s-time-to-cache","title":"Timestep Embedding Tells: It's Time to Cache for Video Diffusion Model","date":"2024-11-28","arxiv_id":"2411.19108","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-vision-diffusion-transformers","slug":"accelerating-vision-diffusion-transformers","title":"Towards Stabilized and Efficient Diffusion Transformers through Long-Skip-Connections with Spectral Constraints","date":"2024-11-26","arxiv_id":"2411.17616","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/accelerating-vision-diffusion-transformers#ran","syntology_url":"https://syntology.ai/paper/2411.17616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17616"}},"official":{"repos":["opensparsellms/skip-dit"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/aigv-assessor-benchmarking-and-evaluating-the","slug":"aigv-assessor-benchmarking-and-evaluating-the","title":"AIGV-Assessor: Benchmarking and Evaluating the Perceptual Quality of Text-to-Video Generation with LMM","date":"2024-11-26","arxiv_id":"2411.17221","repositories_listed":1,"syntology":null},{"url":"/paper/identity-preserving-text-to-video-generation","slug":"identity-preserving-text-to-video-generation","title":"Identity-Preserving Text-to-Video Generation by Frequency Decomposition","date":"2024-11-26","arxiv_id":"2411.17440","repositories_listed":1,"syntology":null},{"url":"/paper/stableanimator-high-quality-identity","slug":"stableanimator-high-quality-identity","title":"StableAnimator: High-Quality Identity-Preserving Human Image Animation","date":"2024-11-26","arxiv_id":"2411.17697","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stableanimator-high-quality-identity#ran","syntology_url":"https://syntology.ai/paper/2411.17697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17697"}},"official":{"repos":["Francis-Rings/StableAnimator"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/intragen-trajectory-controlled-video","slug":"intragen-trajectory-controlled-video","title":"InTraGen: Trajectory-controlled Video Generation for Object Interactions","date":"2024-11-25","arxiv_id":"2411.16804","repositories_listed":1,"syntology":null},{"url":"/paper/moviebench-a-hierarchical-movie-level-dataset","slug":"moviebench-a-hierarchical-movie-level-dataset","title":"MovieBench: A Hierarchical Movie Level Dataset for Long Video Generation","date":"2024-11-22","arxiv_id":"2411.15262","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moviebench-a-hierarchical-movie-level-dataset#ran","syntology_url":"https://syntology.ai/paper/2411.15262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15262"}},"official":{"repos":["showlab/moviebecnh"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neuro-symbolic-evaluation-of-text-to-video","slug":"neuro-symbolic-evaluation-of-text-to-video","title":"Neuro-Symbolic Evaluation of Text-to-Video Models using Formal Verification","date":"2024-11-22","arxiv_id":"2411.16718","repositories_listed":1,"syntology":null},{"url":"/paper/stereocrafter-zero-zero-shot-stereo-video","slug":"stereocrafter-zero-zero-shot-stereo-video","title":"StereoCrafter-Zero: Zero-Shot Stereo Video Generation with Noisy Restart","date":"2024-11-21","arxiv_id":"2411.14295","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":4,"n_honours":1,"n_violates":3,"n_no_contract":4,"n_pointer_only":16,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 3 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/stereocrafter-zero-zero-shot-stereo-video#ran","syntology_url":"https://syntology.ai/paper/2411.14295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14295"}},"official":{"repos":["shijianjian/stereocrafter-zero"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reducio-generating-1024-times-1024-video","slug":"reducio-generating-1024-times-1024-video","title":"REDUCIO! Generating 1024$\\times$1024 Video within 16 Seconds using Extremely Compressed Motion Latents","date":"2024-11-20","arxiv_id":"2411.13552","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reducio-generating-1024-times-1024-video#ran","syntology_url":"https://syntology.ai/paper/2411.13552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13552"}},"official":{"repos":["microsoft/reducio-vae"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vbench-comprehensive-and-versatile-benchmark","slug":"vbench-comprehensive-and-versatile-benchmark","title":"VBench++: Comprehensive and Versatile Benchmark Suite for Video Generative Models","date":"2024-11-20","arxiv_id":"2411.13503","repositories_listed":1,"syntology":null},{"url":"/paper/pom-efficient-image-and-video-generation-with","slug":"pom-efficient-image-and-video-generation-with","title":"PoM: Efficient Image and Video Generation with the Polynomial Mixer","date":"2024-11-19","arxiv_id":"2411.12663","repositories_listed":1,"syntology":null},{"url":"/paper/echomimicv2-towards-striking-simplified-and","slug":"echomimicv2-towards-striking-simplified-and","title":"EchoMimicV2: Towards Striking, Simplified, and Semi-Body Human Animation","date":"2024-11-15","arxiv_id":"2411.10061","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/echomimicv2-towards-striking-simplified-and#ran","syntology_url":"https://syntology.ai/paper/2411.10061","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.10061"}},"official":{"repos":["antgroup/echomimic_v2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/onlyflow-optical-flow-based-motion","slug":"onlyflow-optical-flow-based-motion","title":"OnlyFlow: Optical Flow based Motion Conditioning for Video Diffusion Models","date":"2024-11-15","arxiv_id":"2411.10501","repositories_listed":1,"syntology":null},{"url":"/paper/autoregressive-models-in-vision-a-survey","slug":"autoregressive-models-in-vision-a-survey","title":"Autoregressive Models in Vision: A Survey","date":"2024-11-08","arxiv_id":"2411.05902","repositories_listed":1,"syntology":null},{"url":"/paper/mvsplat360-feed-forward-360-scene-synthesis","slug":"mvsplat360-feed-forward-360-scene-synthesis","title":"MVSplat360: Feed-Forward 360 Scene Synthesis from Sparse Views","date":"2024-11-07","arxiv_id":"2411.04924","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mvsplat360-feed-forward-360-scene-synthesis#ran","syntology_url":"https://syntology.ai/paper/2411.04924","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04924"}},"official":{"repos":["donydchen/mvsplat360"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taming-rectified-flow-for-inversion-and","slug":"taming-rectified-flow-for-inversion-and","title":"Taming Rectified Flow for Inversion and Editing","date":"2024-11-07","arxiv_id":"2411.04746","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/taming-rectified-flow-for-inversion-and#ran","syntology_url":"https://syntology.ai/paper/2411.04746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04746"}},"official":{"repos":["wangjiangshan0725/rf-solver-edit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-and-memory-efficient-video-diffusion","slug":"fast-and-memory-efficient-video-diffusion","title":"Fast and Memory-Efficient Video Diffusion Using Streamlined Inference","date":"2024-11-02","arxiv_id":"2411.01171","repositories_listed":1,"syntology":null},{"url":"/paper/gamegen-x-interactive-open-world-game-video","slug":"gamegen-x-interactive-open-world-game-video","title":"GameGen-X: Interactive Open-world Game Video Generation","date":"2024-11-01","arxiv_id":"2411.00769","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-motion-in-text-to-video-generation","slug":"enhancing-motion-in-text-to-video-generation","title":"Enhancing Motion in Text-to-Video Generation with Decomposed Encoding and Conditioning","date":"2024-10-31","arxiv_id":"2410.24219","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/enhancing-motion-in-text-to-video-generation#ran","syntology_url":"https://syntology.ai/paper/2410.24219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24219"}},"official":{"repos":["pr-ryan/demo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/hellomeme-integrating-spatial-knitting","slug":"hellomeme-integrating-spatial-knitting","title":"HelloMeme: Integrating Spatial Knitting Attentions to Embed High-Level and Fidelity-Rich Conditions in Diffusion Models","date":"2024-10-30","arxiv_id":"2410.22901","repositories_listed":1,"syntology":null},{"url":"/paper/larp-tokenizing-videos-with-a-learned-1","slug":"larp-tokenizing-videos-with-a-learned-1","title":"LARP: Tokenizing Videos with a Learned Autoregressive Generative Prior","date":"2024-10-28","arxiv_id":"2410.21264","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/larp-tokenizing-videos-with-a-learned-1#ran","syntology_url":"https://syntology.ai/paper/2410.21264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21264"}},"official":null}},{"url":"/paper/robust-watermarking-using-generative-priors","slug":"robust-watermarking-using-generative-priors","title":"Robust Watermarking Using Generative Priors Against Image Editing: From Benchmarking to Advances","date":"2024-10-24","arxiv_id":"2410.18775","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-watermarking-using-generative-priors#ran","syntology_url":"https://syntology.ai/paper/2410.18775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18775"}},"official":{"repos":["shilin-lu/vine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/allegro-open-the-black-box-of-commercial","slug":"allegro-open-the-black-box-of-commercial","title":"Allegro: Open the Black Box of Commercial-Level Video Generation Model","date":"2024-10-20","arxiv_id":"2410.15458","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/allegro-open-the-black-box-of-commercial#ran","syntology_url":"https://syntology.ai/paper/2410.15458","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15458"}},"official":{"repos":["rhymes-ai/allegro"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dawn-dynamic-frame-avatar-with-non","slug":"dawn-dynamic-frame-avatar-with-non","title":"DAWN: Dynamic Frame Avatar with Non-autoregressive Diffusion Framework for Talking Head Video Generation","date":"2024-10-17","arxiv_id":"2410.13726","repositories_listed":1,"syntology":null},{"url":"/paper/safree-training-free-and-adaptive-guard-for","slug":"safree-training-free-and-adaptive-guard-for","title":"SAFREE: Training-Free and Adaptive Guard for Safe Text-to-Image And Video Generation","date":"2024-10-16","arxiv_id":"2410.12761","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/safree-training-free-and-adaptive-guard-for#ran","syntology_url":"https://syntology.ai/paper/2410.12761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12761"}},"official":null}},{"url":"/paper/efficient-diffusion-models-a-comprehensive","slug":"efficient-diffusion-models-a-comprehensive","title":"Efficient Diffusion Models: A Comprehensive Survey from Principles to Practices","date":"2024-10-15","arxiv_id":"2410.11795","repositories_listed":1,"syntology":null},{"url":"/paper/lvd-2m-a-long-take-video-dataset-with","slug":"lvd-2m-a-long-take-video-dataset-with","title":"LVD-2M: A Long-take Video Dataset with Temporally Dense Captions","date":"2024-10-14","arxiv_id":"2410.10816","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvd-2m-a-long-take-video-dataset-with#ran","syntology_url":"https://syntology.ai/paper/2410.10816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10816"}},"official":{"repos":["silentview/lvd-2m"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/musetalk-real-time-high-quality-lip","slug":"musetalk-real-time-high-quality-lip","title":"MuseTalk: Real-Time High-Fidelity Video Dubbing via Spatio-Temporal Sampling","date":"2024-10-14","arxiv_id":"2410.10122","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/musetalk-real-time-high-quality-lip#ran","syntology_url":"https://syntology.ai/paper/2410.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10122"}},"official":{"repos":["tmelyralab/musetalk"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tex4d-zero-shot-4d-scene-texturing-with-video","slug":"tex4d-zero-shot-4d-scene-texturing-with-video","title":"Tex4D: Zero-shot 4D Scene Texturing with Video Diffusion Models","date":"2024-10-14","arxiv_id":"2410.10821","repositories_listed":1,"syntology":null},{"url":"/paper/videoagent-self-improving-video-generation","slug":"videoagent-self-improving-video-generation","title":"VideoAgent: Self-Improving Video Generation","date":"2024-10-14","arxiv_id":"2410.10076","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":4,"n_no_contract":6,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 4 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videoagent-self-improving-video-generation#ran","syntology_url":"https://syntology.ai/paper/2410.10076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10076"}},"official":{"repos":["video-as-agent/videoagent"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hallo2-long-duration-and-high-resolution","slug":"hallo2-long-duration-and-high-resolution","title":"Hallo2: Long-Duration and High-Resolution Audio-Driven Portrait Image Animation","date":"2024-10-10","arxiv_id":"2410.07718","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hallo2-long-duration-and-high-resolution#ran","syntology_url":"https://syntology.ai/paper/2410.07718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07718"}},"official":{"repos":["fudan-generative-vision/hallo2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/motionaura-generating-high-quality-and-motion","slug":"motionaura-generating-high-quality-and-motion","title":"MotionAura: Generating High-Quality and Motion Consistent Videos using Discrete Diffusion","date":"2024-10-10","arxiv_id":"2410.07659","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/motionaura-generating-high-quality-and-motion#ran","syntology_url":"https://syntology.ai/paper/2410.07659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07659"}},"official":{"repos":["CandleLabAI/MotionAura-ICLR-2025"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/progressive-autoregressive-video-diffusion","slug":"progressive-autoregressive-video-diffusion","title":"Progressive Autoregressive Video Diffusion Models","date":"2024-10-10","arxiv_id":"2410.08151","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/progressive-autoregressive-video-diffusion#ran","syntology_url":"https://syntology.ai/paper/2410.08151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08151"}},"official":{"repos":["desaixie/pa_vdm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/trans4d-realistic-geometry-aware-transition","slug":"trans4d-realistic-geometry-aware-transition","title":"Trans4D: Realistic Geometry-Aware Transition for Compositional Text-to-4D Synthesis","date":"2024-10-09","arxiv_id":"2410.07155","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trans4d-realistic-geometry-aware-transition#ran","syntology_url":"https://syntology.ai/paper/2410.07155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07155"}},"official":{"repos":["yangling0818/trans4d"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pyramidal-flow-matching-for-efficient-video","slug":"pyramidal-flow-matching-for-efficient-video","title":"Pyramidal Flow Matching for Efficient Video Generative Modeling","date":"2024-10-08","arxiv_id":"2410.05954","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/pyramidal-flow-matching-for-efficient-video#ran","syntology_url":"https://syntology.ai/paper/2410.05954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05954"}},"official":{"repos":["jy0205/Pyramid-Flow"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seeclear-semantic-distillation-enhances-pixel","slug":"seeclear-semantic-distillation-enhances-pixel","title":"SeeClear: Semantic Distillation Enhances Pixel Condensation for Video Super-Resolution","date":"2024-10-08","arxiv_id":"2410.05799","repositories_listed":1,"syntology":null},{"url":"/paper/t2v-turbo-v2-enhancing-video-generation-model","slug":"t2v-turbo-v2-enhancing-video-generation-model","title":"T2V-Turbo-v2: Enhancing Video Generation Model Post-Training through Data, Reward, and Conditional Guidance Design","date":"2024-10-08","arxiv_id":"2410.05677","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/t2v-turbo-v2-enhancing-video-generation-model#ran","syntology_url":"https://syntology.ai/paper/2410.05677","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05677"}},"official":null}},{"url":"/paper/tweediemix-improving-multi-concept-fusion-for","slug":"tweediemix-improving-multi-concept-fusion-for","title":"TweedieMix: Improving Multi-Concept Fusion for Diffusion-based Image/Video Generation","date":"2024-10-08","arxiv_id":"2410.05591","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/tweediemix-improving-multi-concept-fusion-for#ran","syntology_url":"https://syntology.ai/paper/2410.05591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05591"}},"official":{"repos":["kwongihyun/tweediemix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-fvd-enhanced-evaluation-metrics-for","slug":"beyond-fvd-enhanced-evaluation-metrics-for","title":"Beyond FVD: Enhanced Evaluation Metrics for Video Generation Quality","date":"2024-10-07","arxiv_id":"2410.05203","repositories_listed":1,"syntology":null},{"url":"/paper/towards-world-simulator-crafting-physical","slug":"towards-world-simulator-crafting-physical","title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","date":"2024-10-07","arxiv_id":"2410.05363","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-world-simulator-crafting-physical#ran","syntology_url":"https://syntology.ai/paper/2410.05363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05363"}},"official":{"repos":["opengvlab/phygenbench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"b1606d93ae148dc2fe162f6175d0a00c7ff42b547860b57aff2007f786fbc269","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}