{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/audio-generation/papers/2","list_of":"/task/audio-generation","task":"Audio Generation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":3,"rows_per_page":100,"rows":[101,200],"of":270,"counts":{"archive_papers_tagged":270,"with_a_code_link":124,"where_syntology_ran_a_sample":57,"not_listed_spam_title":0,"listed":270,"listed_where_code_ran":57,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":49,"every_run_a_failure_of_syntologys_instrument":8,"listed_with_a_run_with_no_instrument_failure":49,"listed_every_run_a_failure_of_syntologys_instrument":8,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/audio-generation","prev":"/task/audio-generation","next":"/task/audio-generation/papers/3","papers":[{"url":"/paper/sounding-video-generator-a-unified-framework","slug":"sounding-video-generator-a-unified-framework","title":"Sounding Video Generator: A Unified Framework for Text-guided Sounding Video Generation","date":"2023-03-29","arxiv_id":"2303.16541","repositories_listed":1,"syntology":null},{"url":"/paper/av-nerf-learning-neural-fields-for-real-world","slug":"av-nerf-learning-neural-fields-for-real-world","title":"AV-NeRF: Learning Neural Fields for Real-World Audio-Visual Scene Synthesis","date":"2023-02-04","arxiv_id":"2302.02088","repositories_listed":1,"syntology":null},{"url":"/paper/archisound-audio-generation-with-diffusion","slug":"archisound-audio-generation-with-diffusion","title":"ArchiSound: Audio Generation with Diffusion","date":"2023-01-30","arxiv_id":"2301.13267","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/archisound-audio-generation-with-diffusion#ran","syntology_url":"https://syntology.ai/paper/2301.13267","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13267"}},"official":{"repos":["archinetai/audio-diffusion-pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/make-an-audio-text-to-audio-generation-with","slug":"make-an-audio-text-to-audio-generation-with","title":"Make-An-Audio: Text-To-Audio Generation with Prompt-Enhanced Diffusion Models","date":"2023-01-30","arxiv_id":"2301.12661","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/make-an-audio-text-to-audio-generation-with#ran","syntology_url":"https://syntology.ai/paper/2301.12661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12661"}},"official":null}},{"url":"/paper/audiogen-textually-guided-audio-generation","slug":"audiogen-textually-guided-audio-generation","title":"AudioGen: Textually Guided Audio Generation","date":"2022-09-30","arxiv_id":"2209.15352","repositories_listed":1,"syntology":null},{"url":"/paper/diffsound-discrete-diffusion-model-for-text","slug":"diffsound-discrete-diffusion-model-for-text","title":"Diffsound: Discrete Diffusion Model for Text-to-sound Generation","date":"2022-07-20","arxiv_id":"2207.09983","repositories_listed":1,"syntology":null},{"url":"/paper/symphony-generation-with-permutation","slug":"symphony-generation-with-permutation","title":"Symphony Generation with Permutation Invariant Language Model","date":"2022-05-10","arxiv_id":"2205.05448","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/symphony-generation-with-permutation#ran","syntology_url":"https://syntology.ai/paper/2205.05448","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05448"}},"official":{"repos":["symphonynet/SymphonyNet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-source-separation-by-steering","slug":"unsupervised-source-separation-by-steering","title":"Unsupervised Source Separation By Steering Pretrained Music Models","date":"2021-10-25","arxiv_id":"2110.13071","repositories_listed":1,"syntology":null},{"url":"/paper/neural-waveshaping-synthesis","slug":"neural-waveshaping-synthesis","title":"Neural Waveshaping Synthesis","date":"2021-07-11","arxiv_id":"2107.05050","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-waveshaping-synthesis#ran","syntology_url":"https://syntology.ai/paper/2107.05050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.05050"}},"official":{"repos":["ben-hayes/neural-waveshaping-synthesis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/catch-a-waveform-learning-to-generate-audio","slug":"catch-a-waveform-learning-to-generate-audio","title":"Catch-A-Waveform: Learning to Generate Audio from a Single Short Example","date":"2021-06-11","arxiv_id":"2106.06426","repositories_listed":1,"syntology":null},{"url":"/paper/priorgrad-improving-conditional-denoising","slug":"priorgrad-improving-conditional-denoising","title":"PriorGrad: Improving Conditional Denoising Diffusion Models with Data-Dependent Adaptive Prior","date":"2021-06-11","arxiv_id":"2106.06406","repositories_listed":1,"syntology":null},{"url":"/paper/anytime-sampling-for-autoregressive-models-1","slug":"anytime-sampling-for-autoregressive-models-1","title":"Anytime Sampling for Autoregressive Models via Ordered Autoencoding","date":"2021-02-23","arxiv_id":"2102.11495","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/anytime-sampling-for-autoregressive-models-1#ran","syntology_url":"https://syntology.ai/paper/2102.11495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.11495"}},"official":{"repos":["Newbeeer/Anytime-Auto-Regressive-Model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/localize-to-binauralize-audio-spatialization","slug":"localize-to-binauralize-audio-spatialization","title":"Localize to Binauralize: Audio Spatialization From Visual Sound Source Localization","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/phonetic-posteriorgrams-based-many-to-many","slug":"phonetic-posteriorgrams-based-many-to-many","title":"Phonetic Posteriorgrams based Many-to-Many Singing Voice Conversion via Adversarial Training","date":"2020-12-03","arxiv_id":"2012.01837","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/phonetic-posteriorgrams-based-many-to-many#ran","syntology_url":"https://syntology.ai/paper/2012.01837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.01837"}},"official":{"repos":["hhguo/EA-SVC"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audeo-audio-generation-for-a-silent","slug":"audeo-audio-generation-for-a-silent","title":"Audeo: Audio Generation for a Silent Performance Video","date":"2020-06-23","arxiv_id":"2006.14348","repositories_listed":1,"syntology":null},{"url":"/paper/perceiving-music-quality-with-gans","slug":"perceiving-music-quality-with-gans","title":"Perceiving Music Quality with GANs","date":"2020-06-11","arxiv_id":"2006.06287","repositories_listed":1,"syntology":null},{"url":"/paper/unconditional-audio-generation-with","slug":"unconditional-audio-generation-with","title":"Unconditional Audio Generation with Generative Adversarial Networks and Cycle Regularization","date":"2020-05-18","arxiv_id":"2005.08526","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unconditional-audio-generation-with#ran","syntology_url":"https://syntology.ai/paper/2005.08526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.08526"}},"official":{"repos":["ciaua/unagan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/score-and-lyrics-free-singing-voice-1","slug":"score-and-lyrics-free-singing-voice-1","title":"Score and Lyrics-Free Singing Voice Generation","date":"2019-12-26","arxiv_id":"1912.11747","repositories_listed":1,"syntology":null},{"url":"/paper/music-source-separation-in-the-waveform-1","slug":"music-source-separation-in-the-waveform-1","title":"Music Source Separation in the Waveform Domain","date":"2019-11-27","arxiv_id":"1911.13254","repositories_listed":1,"syntology":null},{"url":"/paper/seq-u-net-a-one-dimensional-causal-u-net-for","slug":"seq-u-net-a-one-dimensional-causal-u-net-for","title":"Seq-U-Net: A One-Dimensional Causal U-Net for Efficient Sequence Modelling","date":"2019-11-14","arxiv_id":"1911.06393","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/seq-u-net-a-one-dimensional-causal-u-net-for#ran","syntology_url":"https://syntology.ai/paper/1911.06393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1911.06393"}},"official":{"repos":["f90/Seq-U-Net"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-generation-of-time-frequency","slug":"adversarial-generation-of-time-frequency","title":"Adversarial Generation of Time-Frequency Features with application in audio synthesis","date":"2019-02-11","arxiv_id":"1902.04072","repositories_listed":1,"syntology":null},{"url":"/paper/a-context-encoder-for-audio-inpainting","slug":"a-context-encoder-for-audio-inpainting","title":"Audio inpainting of music by means of neural networks","date":"2018-10-29","arxiv_id":"1810.12138","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-wavegan","slug":"conditional-wavegan","title":"Conditional WaveGAN","date":"2018-09-27","arxiv_id":"1809.10636","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/conditional-wavegan#ran","syntology_url":"https://syntology.ai/paper/1809.10636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.10636"}},"official":{"repos":["acheketa/cwavegan"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/smoothed-dilated-convolutions-for-improved","slug":"smoothed-dilated-convolutions-for-improved","title":"Smoothed Dilated Convolutions for Improved Dense Prediction","date":"2018-08-27","arxiv_id":"1808.08931","repositories_listed":1,"syntology":null},{"url":null,"slug":"freeaudio-training-free-timing-planning-for","title":"FreeAudio: Training-Free Timing Planning for Controllable Long-Form Text-to-Audio Generation","date":"2025-07-11","arxiv_id":"2507.08557","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-by-step-video-to-audio-synthesis-via","title":"Step-by-Step Video-to-Audio Synthesis via Negative Audio Guidance","date":"2025-06-26","arxiv_id":"2506.20995","repositories_listed":0,"syntology":null},{"url":null,"slug":"kling-foley-multimodal-diffusion-transformer","title":"Kling-Foley: Multimodal Diffusion Transformer for High-Quality Video-to-Audio Generation","date":"2025-06-24","arxiv_id":"2506.19774","repositories_listed":0,"syntology":null},{"url":null,"slug":"lilac-a-lightweight-latent-controlnet-for","title":"LiLAC: A Lightweight Latent ControlNet for Musical Audio Generation","date":"2025-06-13","arxiv_id":"2506.11476","repositories_listed":0,"syntology":null},{"url":null,"slug":"visage-video-to-spatial-audio-generation","title":"ViSAGe: Video-to-Spatial Audio Generation","date":"2025-06-13","arxiv_id":"2506.12199","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-automatic-evaluation-methods-on","title":"A Survey of Automatic Evaluation Methods on Text, Visual and Speech Generations","date":"2025-06-06","arxiv_id":"2506.10019","repositories_listed":0,"syntology":null},{"url":null,"slug":"sounding-that-object-interactive-object-aware","title":"Sounding that Object: Interactive Object-Aware Image to Audio Generation","date":"2025-06-04","arxiv_id":"2506.04214","repositories_listed":0,"syntology":null},{"url":null,"slug":"dgmo-training-free-audio-source-separation","title":"DGMO: Training-Free Audio Source Separation through Diffusion-Guided Mask Optimization","date":"2025-06-03","arxiv_id":"2506.02858","repositories_listed":0,"syntology":null},{"url":null,"slug":"infiniteaudio-infinite-length-audio","title":"InfiniteAudio: Infinite-Length Audio Generation with Consistency","date":"2025-06-03","arxiv_id":"2506.03020","repositories_listed":0,"syntology":null},{"url":null,"slug":"impact-iterative-mask-based-parallel-decoding","title":"IMPACT: Iterative Mask-based Parallel Decoding for Text-to-Audio Generation with Diffusion Modeling","date":"2025-05-31","arxiv_id":"2506.00736","repositories_listed":0,"syntology":null},{"url":null,"slug":"audioturbo-fast-text-to-audio-generation-with","title":"AudioTurbo: Fast Text-to-Audio Generation with Rectified Diffusion","date":"2025-05-28","arxiv_id":"2505.22106","repositories_listed":0,"syntology":null},{"url":null,"slug":"envsdd-benchmarking-environmental-sound","title":"EnvSDD: Benchmarking Environmental Sound Deepfake Detection","date":"2025-05-25","arxiv_id":"2505.19203","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmmg-a-comprehensive-and-reliable-evaluation","title":"MMMG: a Comprehensive and Reliable Evaluation Suite for Multitask Multimodal Generation","date":"2025-05-23","arxiv_id":"2505.17613","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-cross-modal-translation-of-score","title":"Unified Cross-modal Translation of Score Images, Symbolic Music, and Performance Audio","date":"2025-05-19","arxiv_id":"2505.12863","repositories_listed":0,"syntology":null},{"url":null,"slug":"dpn-gan-inducing-periodic-activations-in","title":"DPN-GAN: Inducing Periodic Activations in Generative Adversarial Networks for High-Fidelity Audio Synthesis","date":"2025-05-14","arxiv_id":"2505.09091","repositories_listed":0,"syntology":null},{"url":null,"slug":"tacos-temporally-aligned-audio-captions-for","title":"TACOS: Temporally-aligned Audio CaptiOnS for Language-Audio Pretraining","date":"2025-05-12","arxiv_id":"2505.07609","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-optimal-transport-and-voice","title":"Discrete Optimal Transport and Voice Conversion","date":"2025-05-07","arxiv_id":"2505.04382","repositories_listed":0,"syntology":null},{"url":null,"slug":"wasserstein-convergence-of-score-based","title":"Wasserstein Convergence of Score-based Generative Models under Semiconvexity and Discontinuous Gradients","date":"2025-05-06","arxiv_id":"2505.03432","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-design-of-diffusion-based-neural","title":"On the Design of Diffusion-based Neural Speech Codecs","date":"2025-04-11","arxiv_id":"2504.08470","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10005","title":"Multimodal Cinematic Video Synthesis Using Text-to-Image and Audio Generation Models","date":"2025-04-06","arxiv_id":"2506.10005","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepaudio-v1-towards-multi-modal-multi-stage","title":"DeepAudio-V1:Towards Multi-Modal Multi-Stage End-to-End Video to Speech and Audio Generation","date":"2025-03-28","arxiv_id":"2503.22265","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepsound-v1-start-to-think-step-by-step-in","title":"DeepSound-V1: Start to Think Step-by-Step in the Audio Generation from Videos","date":"2025-03-28","arxiv_id":"2503.22208","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-some-noise-towards-llm-audio-reasoning","title":"Make Some Noise: Towards LLM audio reasoning and generation using sound tokens","date":"2025-03-28","arxiv_id":"2503.22275","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffgap-a-lightweight-diffusion-module-in","title":"DiffGAP: A Lightweight Diffusion Module in Contrastive Space for Bridging Cross-Model Gap","date":"2025-03-15","arxiv_id":"2503.12131","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiox-diffusion-transformer-for-anything-to","title":"AudioX: Diffusion Transformer for Anything-to-Audio Generation","date":"2025-03-13","arxiv_id":"2503.10522","repositories_listed":0,"syntology":null},{"url":null,"slug":"ta-v2a-textually-assisted-video-to-audio","title":"TA-V2A: Textually Assisted Video-to-Audio Generation","date":"2025-03-12","arxiv_id":"2503.10700","repositories_listed":0,"syntology":null},{"url":null,"slug":"reelwave-a-multi-agent-framework-toward","title":"ReelWave: Multi-Agentic Movie Sound Generation through Multimodal LLM Conversation","date":"2025-03-10","arxiv_id":"2503.07217","repositories_listed":0,"syntology":null},{"url":null,"slug":"synchronized-video-to-audio-generation-via","title":"Synchronized Video-to-Audio Generation via Mel Quantization-Continuum Decomposition","date":"2025-03-10","arxiv_id":"2503.06984","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-audio-generation-from-dynamic-mri-via","title":"Speech Audio Generation from dynamic MRI via a Knowledge Enhanced Conditional Variational Autoencoder","date":"2025-03-09","arxiv_id":"2503.06588","repositories_listed":0,"syntology":null},{"url":null,"slug":"dualspec-text-to-spatial-audio-generation-via","title":"DualSpec: Text-to-spatial-audio Generation via Dual-Spectrogram Guided Diffusion Model","date":"2025-02-26","arxiv_id":"2502.18952","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-quantum-algorithms-for","title":"Towards efficient quantum algorithms for diffusion probability models","date":"2025-02-20","arxiv_id":"2502.14252","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiospa-spatializing-sound-events-with-text","title":"AudioSpa: Spatializing Sound Events with Text","date":"2025-02-16","arxiv_id":"2502.11219","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-a-unified-diffusion-transformer-for","title":"UniForm: A Unified Multi-Task Diffusion Transformer for Audio-Video Generation","date":"2025-02-06","arxiv_id":"2502.03897","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosyaudio-improving-audio-generation-with","title":"CosyAudio: Improving Audio Generation with Confidence Scores and Synthetic Captions","date":"2025-01-28","arxiv_id":"2501.16761","repositories_listed":0,"syntology":null},{"url":null,"slug":"ccstereo-audio-visual-contextual-and","title":"CCStereo: Audio-Visual Contextual and Contrastive Learning for Binaural Audio Generation","date":"2025-01-06","arxiv_id":"2501.02786","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonparametric-estimation-of-a-factorizable","title":"Nonparametric estimation of a factorizable density using diffusion models","date":"2025-01-03","arxiv_id":"2501.01783","repositories_listed":0,"syntology":null},{"url":null,"slug":"animate-and-sound-an-image","title":"Animate and Sound an Image","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"foley-flow-coordinated-video-to-audio","title":"Foley-Flow: Coordinated Video-to-Audio Generation with Masked Audio-Visual Alignment and Dynamic Conditional Flows","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tri-ergon-fine-grained-video-to-audio","title":"Tri-Ergon: Fine-grained Video-to-Audio Generation with Multi-modal Conditions and LUFS Control","date":"2024-12-29","arxiv_id":"2412.20378","repositories_listed":0,"syntology":null},{"url":null,"slug":"voicedit-dual-condition-diffusion-transformer","title":"VoiceDiT: Dual-Condition Diffusion Transformer for Environment-Aware Speech Synthesis","date":"2024-12-26","arxiv_id":"2412.19259","repositories_listed":0,"syntology":null},{"url":null,"slug":"smooth-foley-creating-continuous-sound-for","title":"Smooth-Foley: Creating Continuous Sound for Video-to-Audio Generation Under Semantic Guidance","date":"2024-12-24","arxiv_id":"2412.18157","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-v2a-synthesis-of-synchronized-sound","title":"FolAI: Synchronized Foley Sound Generation with Semantic and Temporal Alignment","date":"2024-12-19","arxiv_id":"2412.15023","repositories_listed":0,"syntology":null},{"url":null,"slug":"vintage-joint-video-and-text-conditioning-for","title":"VinTAGe: Joint Video and Text Conditioning for Holistic Audio Generation","date":"2024-12-14","arxiv_id":"2412.10768","repositories_listed":0,"syntology":null},{"url":null,"slug":"yingsound-video-guided-sound-effects","title":"YingSound: Video-Guided Sound Effects Generation with Multi-modal Chain-of-Thought Controls","date":"2024-12-12","arxiv_id":"2412.09168","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-audio-query-handling-system","title":"Comprehensive Audio Query Handling System with Integrated Expert Models and Contextual Understanding","date":"2024-12-05","arxiv_id":"2412.03980","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-autoregressive-models-with-noise","title":"Continuous Autoregressive Models with Noise Augmentation Avoid Error Accumulation","date":"2024-11-27","arxiv_id":"2411.18447","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-guided-foley-sound-generation-with","title":"Video-Guided Foley Sound Generation with Multimodal Controls","date":"2024-11-26","arxiv_id":"2411.17698","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceivers-a-multi-scale-perceiver-with","title":"PerceiverS: A Multi-Scale Perceiver with Effective Segmentation for Long-Term Expressive Symbolic Music Generation","date":"2024-11-13","arxiv_id":"2411.08307","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiobox-tta-rag-improving-zero-shot-and-few","title":"Audiobox TTA-RAG: Improving Zero-Shot and Few-Shot Text-To-Audio with Retrieval-Augmented Generation","date":"2024-11-07","arxiv_id":"2411.05141","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-npu-hwc-system-for-the-iscslp-2024","title":"The NPU-HWC System for the ISCSLP 2024 Inspirational and Convincing Audio Generation Challenge","date":"2024-10-31","arxiv_id":"2410.23815","repositories_listed":0,"syntology":null},{"url":null,"slug":"both-ears-wide-open-towards-language-driven","title":"Both Ears Wide Open: Towards Language-Driven Spatial Audio Generation","date":"2024-10-14","arxiv_id":"2410.10676","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-agent-leveraging-llms-for-audio","title":"Audio-Agent: Leveraging LLMs For Audio Generation, Editing and Composition","date":"2024-10-04","arxiv_id":"2410.03335","repositories_listed":0,"syntology":null},{"url":null,"slug":"did-you-hear-that-introducing-aadg-a","title":"Did You Hear That? Introducing AADG: A Framework for Generating Benchmark Data in Audio Anomaly Detection","date":"2024-10-04","arxiv_id":"2410.03904","repositories_listed":0,"syntology":null},{"url":null,"slug":"mdsgen-fast-and-efficient-masked-diffusion","title":"MDSGen: Fast and Efficient Masked Diffusion Temporal-Aware Transformers for Open-Domain Sound Generation","date":"2024-10-03","arxiv_id":"2410.02130","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-and-mitigating-inconsistency-in","title":"Analyzing and Mitigating Inconsistency in Discrete Audio Tokens for Neural Codec Language Models","date":"2024-09-28","arxiv_id":"2409.19283","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-to-audio-generation-with-fine-grained","title":"Video-to-Audio Generation with Fine-grained Temporal Semantics","date":"2024-09-23","arxiv_id":"2409.14709","repositories_listed":0,"syntology":null},{"url":null,"slug":"ptq4adm-post-training-quantization-for","title":"PTQ4ADM: Post-Training Quantization for Efficient Text Conditional Audio Diffusion Models","date":"2024-09-20","arxiv_id":"2409.13894","repositories_listed":0,"syntology":null},{"url":null,"slug":"audiocomposer-towards-fine-grained-audio","title":"AudioComposer: Towards Fine-grained Audio Generation with Natural Language Descriptions","date":"2024-09-19","arxiv_id":"2409.12560","repositories_listed":0,"syntology":null},{"url":null,"slug":"ndvq-robust-neural-audio-codec-with-normal","title":"NDVQ: Robust Neural Audio Codec with Normal Distribution-Based Vector Quantization","date":"2024-09-19","arxiv_id":"2409.12717","repositories_listed":0,"syntology":null},{"url":null,"slug":"ezaudio-enhancing-text-to-audio-generation","title":"EzAudio: Enhancing Text-to-Audio Generation with Efficient Diffusion Transformer","date":"2024-09-17","arxiv_id":"2409.10819","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-source-disentanglement-in-neural","title":"Learning Source Disentanglement in Neural Audio Codec","date":"2024-09-17","arxiv_id":"2409.11228","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-prompt-is-not-enough-sound-event","title":"Text Prompt is Not Enough: Sound Event Enhanced Prompt Adapter for Target Style Audio Generation","date":"2024-09-14","arxiv_id":"2409.09381","repositories_listed":0,"syntology":null},{"url":null,"slug":"metabgm-dynamic-soundtrack-transformation-for","title":"MetaBGM: Dynamic Soundtrack Transformation For Continuous Multi-Scene Experiences With Ambient Awareness And Personalization","date":"2024-09-05","arxiv_id":"2409.03844","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-and-advances-of-artificial","title":"Applications and Advances of Artificial Intelligence in Music Generation:A Review","date":"2024-09-03","arxiv_id":"2409.03715","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-foley-two-stage-video-to-sound","title":"Video-Foley: Two-Stage Video-To-Sound Generation via Temporal Event Condition For Foley Sound","date":"2024-08-21","arxiv_id":"2408.11915","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-the-communication","title":"Demystifying the Communication Characteristics for Distributed Transformer Models","date":"2024-08-19","arxiv_id":"2408.10197","repositories_listed":0,"syntology":null},{"url":null,"slug":"connective-viewpoints-of-signal-to-noise","title":"Connective Viewpoints of Signal-to-Noise Diffusion Models","date":"2024-08-08","arxiv_id":"2408.04221","repositories_listed":0,"syntology":null},{"url":null,"slug":"braille-to-speech-generator-audio-generation","title":"Braille-to-Speech Generator: Audio Generation Based on Joint Fine-Tuning of CLIP and Fastspeech2","date":"2024-07-19","arxiv_id":"2407.14212","repositories_listed":0,"syntology":null},{"url":null,"slug":"medic-zero-shot-music-editing-with","title":"MEDIC: Zero-shot Music Editing with Disentangled Inversion Control","date":"2024-07-18","arxiv_id":"2407.13220","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-and-driving-human-body-soundfields","title":"Modeling and Driving Human Body Soundfields through Acoustic Primitives","date":"2024-07-18","arxiv_id":"2407.13083","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-to-audio-generation-with-hidden","title":"Video-to-Audio Generation with Hidden Alignment","date":"2024-07-10","arxiv_id":"2407.07464","repositories_listed":0,"syntology":null},{"url":null,"slug":"soaf-scene-occlusion-aware-neural-acoustic","title":"SOAF: Scene Occlusion-aware Neural Acoustic Field","date":"2024-07-02","arxiv_id":"2407.02264","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-statistical-rates-for-consistency","title":"Provable Statistical Rates for Consistency Diffusion Models","date":"2024-06-23","arxiv_id":"2406.16213","repositories_listed":0,"syntology":null},{"url":null,"slug":"action2sound-ambient-aware-generation-of","title":"Action2Sound: Ambient-Aware Generation of Action Sounds from Egocentric Videos","date":"2024-06-13","arxiv_id":"2406.09272","repositories_listed":0,"syntology":null},{"url":null,"slug":"codecfake-an-initial-dataset-for-detecting","title":"Codecfake: An Initial Dataset for Detecting LLM-based Deepfake Audio","date":"2024-06-12","arxiv_id":"2406.08112","repositories_listed":0,"syntology":null},{"url":null,"slug":"av-dit-efficient-audio-visual-diffusion","title":"AV-DiT: Efficient Audio-Visual Diffusion Transformer for Joint Audio and Video Generation","date":"2024-06-11","arxiv_id":"2406.07686","repositories_listed":0,"syntology":null}],"record_sha256":"fa2cb93063f97e9a570531fa5ee9257ba0baa0644fe089ad52d6dd2ef1388fd3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}