{"url":"/task/scene-generation","name":"Scene Generation","slug":"scene-generation","description_markdown":"make to t shirt an Ad with a little bit of action","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":309,"papers_with_code":122,"benchmarks":6,"benchmark_tables_in_archive":6,"benchmark_tables_shown":6,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":9,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/scene-generation-on-googleearth","slug":"scene-generation-on-googleearth","dataset":"GoogleEarth","dataset_url":"/dataset/googleearth","rows_in_archive":5,"metrics":["Depth Error","KID","Camera Error","FID"],"first_row_in_archive_order":{"model":"GaussianCity","paper_title":"GaussianCity: Generative Gaussian Splatting for Unbounded 3D City Generation","paper_url":"/paper/gaussiancity-generative-gaussian-splatting","paper_date":"2024-06-10","arxiv_id":"2406.06526","code_links":[{"title":"hzxie/GaussianCity","url":"https://github.com/hzxie/GaussianCity"}],"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":10}}},{"leaderboard":"/sota/scene-generation-on-avd","slug":"scene-generation-on-avd","dataset":"AVD","dataset_url":"/dataset/avd","rows_in_archive":3,"metrics":["FID","SwAV-FID"],"first_row_in_archive_order":{"model":"GSN","paper_title":"Unconstrained Scene Generation with Locally Conditioned Radiance Fields","paper_url":"/paper/unconstrained-scene-generation-with-locally","paper_date":"2021-04-01","arxiv_id":"2104.00670","code_links":[{"title":"apple/ml-gsn","url":"https://github.com/apple/ml-gsn"}],"syntology":null}},{"leaderboard":"/sota/scene-generation-on-replica","slug":"scene-generation-on-replica","dataset":"Replica","dataset_url":"/dataset/replica","rows_in_archive":3,"metrics":["FID","SwAV-FID"],"first_row_in_archive_order":{"model":"GSN","paper_title":"Unconstrained Scene Generation with Locally Conditioned Radiance Fields","paper_url":"/paper/unconstrained-scene-generation-with-locally","paper_date":"2021-04-01","arxiv_id":"2104.00670","code_links":[{"title":"apple/ml-gsn","url":"https://github.com/apple/ml-gsn"}],"syntology":null}},{"leaderboard":"/sota/scene-generation-on-vizdoom","slug":"scene-generation-on-vizdoom","dataset":"VizDoom","dataset_url":"/dataset/vizdoom","rows_in_archive":3,"metrics":["FID","SwAV-FID"],"first_row_in_archive_order":{"model":"GSN","paper_title":"Unconstrained Scene Generation with Locally Conditioned Radiance Fields","paper_url":"/paper/unconstrained-scene-generation-with-locally","paper_date":"2021-04-01","arxiv_id":"2104.00670","code_links":[{"title":"apple/ml-gsn","url":"https://github.com/apple/ml-gsn"}],"syntology":null}},{"leaderboard":"/sota/scene-generation-on-osm","slug":"scene-generation-on-osm","dataset":"OSM","dataset_url":"/dataset/osm","rows_in_archive":2,"metrics":["Average FID","KID"],"first_row_in_archive_order":{"model":"InfiniteGAN","paper_title":"InfinityGAN: Towards Infinite-Pixel Image Synthesis","paper_url":"/paper/infinitygan-towards-infinite-resolution-image","paper_date":"2021-04-08","arxiv_id":"2104.03963","code_links":[{"title":"hubert0527/infinityGAN","url":"https://github.com/hubert0527/infinityGAN"}],"syntology":{"n":14,"n_ran":7,"n_unverified":7,"n_pointer_only":14}}},{"leaderboard":"/sota/scene-generation-on-kitti","slug":"scene-generation-on-kitti","dataset":"KITTI","dataset_url":"/dataset/kitti","rows_in_archive":1,"metrics":["FID","KID"],"first_row_in_archive_order":{"model":"GaussianCity","paper_title":"GaussianCity: Generative Gaussian Splatting for Unbounded 3D City Generation","paper_url":"/paper/gaussiancity-generative-gaussian-splatting","paper_date":"2024-06-10","arxiv_id":"2406.06526","code_links":[{"title":"hzxie/GaussianCity","url":"https://github.com/hzxie/GaussianCity"}],"syntology":{"n":10,"n_ran":7,"n_unverified":3,"n_pointer_only":10}}}],"datasets":[{"url":"/dataset/kitti","name":"KITTI","full_name":"","num_papers_in_archive":3661},{"url":"/dataset/replica","name":"Replica","full_name":"","num_papers_in_archive":414},{"url":"/dataset/vizdoom","name":"VizDoom","full_name":"VizDoom","num_papers_in_archive":156},{"url":"/dataset/avd","name":"AVD","full_name":"Active Vision Dataset","num_papers_in_archive":29},{"url":"/dataset/codraw","name":"CoDraw","full_name":null,"num_papers_in_archive":14},{"url":"/dataset/osm","name":"OSM","full_name":"","num_papers_in_archive":8},{"url":"/dataset/googleearth","name":"GoogleEarth","full_name":"","num_papers_in_archive":5},{"url":"/dataset/3d-front-human","name":"3D FRONT HUMAN","full_name":"","num_papers_in_archive":2},{"url":"/dataset/instaorder","name":"InstaOrder","full_name":"","num_papers_in_archive":2}],"subtasks":[],"parent_tasks":[{"url":"/task/16k","name":"16k"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":122,"tagged_in_all":309,"items":[{"url":"/paper/funnel-activation-for-visual-recognition","title":"Funnel Activation for Visual Recognition","date":"2020-07-23","arxiv_id":"2007.11824","repositories_listed":7,"syntology":null},{"url":"/paper/luminous-indoor-scene-generation-for-embodied","title":"LUMINOUS: Indoor Scene Generation for Embodied AI Challenges","date":"2021-11-10","arxiv_id":"2111.05527","repositories_listed":4,"syntology":null},{"url":"/paper/root-vlm-based-system-for-indoor-scene","title":"ROOT: VLM based System for Indoor Scene Understanding and Beyond","date":"2024-11-24","arxiv_id":"2411.15714","repositories_listed":3,"syntology":null},{"url":"/paper/pi-gan-periodic-implicit-generative","title":"pi-GAN: Periodic Implicit Generative Adversarial Networks for 3D-Aware Image Synthesis","date":"2020-12-02","arxiv_id":"2012.00926","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/gpd-1-generative-pre-training-for-driving","title":"GPD-1: Generative Pre-training for Driving","date":"2024-12-11","arxiv_id":"2412.08643","repositories_listed":2,"syntology":null},{"url":"/paper/laion-sg-an-enhanced-large-scale-dataset-for","title":"LAION-SG: An Enhanced Large-Scale Dataset for Training Complex Image-Text Models with Structural Annotations","date":"2024-12-11","arxiv_id":"2412.08580","repositories_listed":2,"syntology":{"n":23,"n_ran":6,"n_unverified":17,"n_pointer_only":12}},{"url":"/paper/alfie-democratising-rgba-image-generation","title":"Alfie: Democratising RGBA Image Generation With No $$$","date":"2024-08-27","arxiv_id":"2408.14826","repositories_listed":2,"syntology":null},{"url":"/paper/riskawarebench-towards-evaluating-physical","title":"EARBench: Towards Evaluating Physical Risk Awareness for Task Planning of Foundation Model-based Embodied AI Agents","date":"2024-08-08","arxiv_id":"2408.04449","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/pyramid-diffusion-for-fine-3d-large-scene","title":"Pyramid Diffusion for Fine 3D Large Scene Generation","date":"2023-11-20","arxiv_id":"2311.12085","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/genesis-v2-inferring-unordered-object","title":"GENESIS-V2: Inferring Unordered Object Representations without Iterative Refinement","date":"2021-04-20","arxiv_id":"2104.09958","repositories_listed":2,"syntology":null},{"url":"/paper/generative-adversarial-transformers","title":"Generative Adversarial Transformers","date":"2021-03-01","arxiv_id":"2103.01209","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/sceneformer-indoor-scene-generation-with","title":"SceneFormer: Indoor Scene Generation with Transformers","date":"2020-12-17","arxiv_id":"2012.09793","repositories_listed":2,"syntology":null},{"url":"/paper/learning-object-placements-for-relational","title":"Learning Object Placements For Relational Instructions by Hallucinating Scene Representations","date":"2020-01-23","arxiv_id":"2001.08481","repositories_listed":2,"syntology":null},{"url":"/paper/local-class-specific-and-global-image-level","title":"Local Class-Specific and Global Image-Level Generative Adversarial Networks for Semantic-Guided Scene Generation","date":"2019-12-27","arxiv_id":"1912.12215","repositories_listed":2,"syntology":null},{"url":"/paper/learning-canonical-representations-for-scene","title":"Learning Canonical Representations for Scene Graph to Image Generation","date":"2019-12-16","arxiv_id":"1912.07414","repositories_listed":2,"syntology":{"n":17,"n_ran":2,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/semantic-bottleneck-scene-generation","title":"Semantic Bottleneck Scene Generation","date":"2019-11-26","arxiv_id":"1911.11357","repositories_listed":2,"syntology":null},{"url":"/paper/specifying-object-attributes-and-relations-in","title":"Specifying Object Attributes and Relations in Interactive Scene Generation","date":"2019-09-11","arxiv_id":"1909.05379","repositories_listed":2,"syntology":{"n":16,"n_ran":0,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/genesis-generative-scene-inference-and","title":"GENESIS: Generative Scene Inference and Sampling with Object-Centric Latent Representations","date":"2019-07-30","arxiv_id":"1907.13052","repositories_listed":2,"syntology":null},{"url":"/paper/scenegraphnet-neural-message-passing-for-3d","title":"SceneGraphNet: Neural Message Passing for 3D Indoor Scene Augmentation","date":"2019-07-25","arxiv_id":"1907.11308","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/layoutvae-stochastic-scene-layout-generation","title":"LayoutVAE: Stochastic Scene Layout Generation From a Label Set","date":"2019-07-24","arxiv_id":"1907.10719","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/scenic-language-based-scene-generation","title":"Scenic: A Language for Scenario Specification and Scene Generation","date":"2018-09-25","arxiv_id":"1809.09310","repositories_listed":2,"syntology":null},{"url":"/paper/i-2-world-intra-inter-tokenization-for","title":"$I^{2}$-World: Intra-Inter Tokenization for Efficient Dynamic 4D Scene Forecasting","date":"2025-07-12","arxiv_id":"2507.09144","repositories_listed":1,"syntology":null},{"url":"/paper/xverse-consistent-multi-subject-control-of","title":"XVerse: Consistent Multi-Subject Control of Identity and Semantic Attributes via DiT Modulation","date":"2025-06-26","arxiv_id":"2506.21416","repositories_listed":1,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/long-term-traffic-simulation-with-interleaved","title":"Long-term Traffic Simulation with Interleaved Autoregressive Motion and Scenario Generation","date":"2025-06-20","arxiv_id":"2506.17213","repositories_listed":1,"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/3d-scene-generation-a-survey","title":"3D Scene Generation: A Survey","date":"2025-05-08","arxiv_id":"2505.05474","repositories_listed":1,"syntology":null},{"url":"/paper/steerable-scene-generation-with-post-training","title":"Steerable Scene Generation with Post Training and Inference-Time Search","date":"2025-05-07","arxiv_id":"2505.04831","repositories_listed":1,"syntology":null},{"url":"/paper/dualdiff-dual-branch-diffusion-model-for","title":"DualDiff: Dual-branch Diffusion Model for Autonomous Driving with Semantic Fusion","date":"2025-05-03","arxiv_id":"2505.01857","repositories_listed":1,"syntology":null},{"url":"/paper/holotime-taming-video-diffusion-models-for","title":"HoloTime: Taming Video Diffusion Models for Panoramic 4D Scene Generation","date":"2025-04-30","arxiv_id":"2504.21650","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/ep-diffuser-an-efficient-diffusion-model-for","title":"EP-Diffuser: An Efficient Diffusion Model for Traffic Scene Generation and Prediction via Polynomial Representations","date":"2025-04-07","arxiv_id":"2504.05422","repositories_listed":1,"syntology":null},{"url":"/paper/free4d-tuning-free-4d-scene-generation-with","title":"Free4D: Tuning-free 4D Scene Generation with Spatial-Temporal Consistency","date":"2025-03-26","arxiv_id":"2503.20785","repositories_listed":1,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}