{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/vq-vae/papers/2","list_of":"/method/vq-vae","method":"VQ-VAE","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,197],"of":197,"counts":{"archive_papers_tagged":197,"with_a_code_link":82,"where_syntology_ran_a_sample":35,"not_listed_spam_title":0,"listed":197,"listed_where_code_ran":35,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":32,"every_run_a_failure_of_syntologys_instrument":3,"listed_with_a_run_with_no_instrument_failure":32,"listed_every_run_a_failure_of_syntologys_instrument":3,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/vq-vae","prev":"/method/vq-vae","next":null,"papers":[{"paper":null,"slug":"layoutdiffuse-adapting-foundational-diffusion","title":"LayoutDiffuse: Adapting Foundational Diffusion Models for Layout-to-Image Generation","date":"2023-02-16","arxiv_id":"2302.08908","n_code_links":0,"syntology":null},{"paper":null,"slug":"vector-quantized-wasserstein-auto-encoder","title":"Vector Quantized Wasserstein Auto-Encoder","date":"2023-02-12","arxiv_id":"2302.05917","n_code_links":0,"syntology":null},{"paper":"/paper/language-quantized-autoencoders-towards-1","slug":"language-quantized-autoencoders-towards-1","title":"Language Quantized AutoEncoders: Towards Unsupervised Text-Image Alignment","date":"2023-02-02","arxiv_id":"2302.00902","n_code_links":1,"syntology":null},{"paper":"/paper/improving-statistical-fidelity-for-neural","slug":"improving-statistical-fidelity-for-neural","title":"Improving Statistical Fidelity for Neural Image Compression with Implicit Local Likelihood Models","date":"2023-01-26","arxiv_id":"2301.11189","n_code_links":1,"syntology":null},{"paper":"/paper/t2m-gpt-generating-human-motion-from-textual","slug":"t2m-gpt-generating-human-motion-from-textual","title":"T2M-GPT: Generating Human Motion from Textual Descriptions with Discrete Representations","date":"2023-01-15","arxiv_id":"2301.06052","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Mael-zys/T2M-GPT"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/latent-autoregressive-source-separation","slug":"latent-autoregressive-source-separation","title":"Latent Autoregressive Source Separation","date":"2023-01-09","arxiv_id":"2301.08562","n_code_links":1,"syntology":null},{"paper":"/paper/codetalker-speech-driven-3d-facial-animation","slug":"codetalker-speech-driven-3d-facial-animation","title":"CodeTalker: Speech-Driven 3D Facial Animation with Discrete Motion Prior","date":"2023-01-06","arxiv_id":"2301.02379","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":8,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["Doubiiu/CodeTalker"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/all-in-tokens-unifying-output-space-of-visual","slug":"all-in-tokens-unifying-output-space-of-visual","title":"All in Tokens: Unifying Output Space of Visual Tasks via Soft Token","date":"2023-01-05","arxiv_id":"2301.02229","n_code_links":1,"syntology":null},{"paper":null,"slug":"generating-human-motion-from-textual","title":"Generating Human Motion From Textual Descriptions With Discrete Representations","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"image-compression-with-product-quantized","title":"Image Compression with Product Quantized Masked Image Modeling","date":"2022-12-14","arxiv_id":"2212.07372","n_code_links":0,"syntology":null},{"paper":"/paper/generating-holistic-3d-human-motion-from","slug":"generating-holistic-3d-human-motion-from","title":"Generating Holistic 3D Human Motion from Speech","date":"2022-12-08","arxiv_id":"2212.04420","n_code_links":3,"syntology":null},{"paper":null,"slug":"map-music2vec-a-simple-and-effective-baseline","title":"MAP-Music2Vec: A Simple and Effective Baseline for Self-Supervised Music Audio Representation Learning","date":"2022-12-05","arxiv_id":"2212.02508","n_code_links":0,"syntology":null},{"paper":"/paper/melody-transcription-via-generative-pre","slug":"melody-transcription-via-generative-pre","title":"Melody transcription via generative pre-training","date":"2022-12-04","arxiv_id":"2212.01884","n_code_links":1,"syntology":null},{"paper":"/paper/accelerating-antimicrobial-peptide-discovery","slug":"accelerating-antimicrobial-peptide-discovery","title":"Accelerating Antimicrobial Peptide Discovery with Latent Structure","date":"2022-11-28","arxiv_id":"2212.09450","n_code_links":1,"syntology":null},{"paper":"/paper/homology-constrained-vector-quantization","slug":"homology-constrained-vector-quantization","title":"Homology-constrained vector quantization entropy regularizer","date":"2022-11-25","arxiv_id":"2211.14363","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["st4cks1defl0w/hcvq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"disentangled-feature-learning-for-real-time","title":"Disentangled Feature Learning for Real-Time Neural Speech Coding","date":"2022-11-22","arxiv_id":"2211.11960","n_code_links":0,"syntology":null},{"paper":"/paper/edge-editable-dance-generation-from-music","slug":"edge-editable-dance-generation-from-music","title":"EDGE: Editable Dance Generation From Music","date":"2022-11-19","arxiv_id":"2211.10658","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":8,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 3 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Stanford-TML/EDGE"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ssgvs-semantic-scene-graph-to-video-synthesis","title":"SSGVS: Semantic Scene Graph-to-Video Synthesis","date":"2022-11-11","arxiv_id":"2211.06119","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-control-by-iterative-inversion","title":"Learning Control by Iterative Inversion","date":"2022-11-03","arxiv_id":"2211.01724","n_code_links":0,"syntology":null},{"paper":"/paper/character-centric-story-visualization-via","slug":"character-centric-story-visualization-via","title":"Character-Centric Story Visualization via Visual Planning and Token Alignment","date":"2022-10-16","arxiv_id":"2210.08465","n_code_links":2,"syntology":{"ran":12,"of":21,"n_ran_checked":9,"n_instrument":3,"unverified":9,"pointer_only":0,"phrase":"12 ran (of which 7 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","official":{"repos":["pluslabnlp/vp-csv","sairin1202/vp-csv"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/jukedrummer-conditional-beat-aware-audio","slug":"jukedrummer-conditional-beat-aware-audio","title":"JukeDrummer: Conditional Beat-aware Audio-domain Drum Accompaniment Generation via Transformer VQ-VAE","date":"2022-10-12","arxiv_id":"2210.06007","n_code_links":1,"syntology":null},{"paper":null,"slug":"race-bias-analysis-of-bona-fide-errors-in","title":"Race Bias Analysis of Bona Fide Errors in face anti-spoofing","date":"2022-10-11","arxiv_id":"2210.05366","n_code_links":0,"syntology":null},{"paper":"/paper/lmqformer-a-laplace-prior-guided-mask-query","slug":"lmqformer-a-laplace-prior-guided-mask-query","title":"LMQFormer: A Laplace-Prior-Guided Mask Query Transformer for Lightweight Snow Removal","date":"2022-10-10","arxiv_id":"2210.04787","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-deep-learning-approach-for-detection-and","title":"A deep learning approach for detection and localization of leaf anomalies","date":"2022-10-07","arxiv_id":"2210.03558","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-planning-in-a-compact-latent-action","slug":"efficient-planning-in-a-compact-latent-action","title":"Efficient Planning in a Compact Latent Action Space","date":"2022-08-22","arxiv_id":"2208.10291","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ZhengyaoJiang/latentplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/hierarchical-residual-learning-based-vector","slug":"hierarchical-residual-learning-based-vector","title":"Hierarchical Residual Learning Based Vector Quantized Variational Autoencoder for Image Reconstruction and Generation","date":"2022-08-09","arxiv_id":"2208.04554","n_code_links":1,"syntology":null},{"paper":null,"slug":"augmenting-vision-language-pretraining-by","title":"Augmenting Vision Language Pretraining by Learning Codebook with Visual Semantics","date":"2022-07-31","arxiv_id":"2208.00475","n_code_links":0,"syntology":null},{"paper":null,"slug":"cancer-subtyping-by-improved-transcriptomic","title":"Cancer Subtyping by Improved Transcriptomic Features Using Vector Quantized Variational Autoencoder","date":"2022-07-20","arxiv_id":"2207.09783","n_code_links":0,"syntology":null},{"paper":"/paper/diffsound-discrete-diffusion-model-for-text","slug":"diffsound-discrete-diffusion-model-for-text","title":"Diffsound: Discrete Diffusion Model for Text-to-sound Generation","date":"2022-07-20","arxiv_id":"2207.09983","n_code_links":1,"syntology":null},{"paper":null,"slug":"predictive-neural-speech-coding","title":"Latent-Domain Predictive Neural Speech Coding","date":"2022-07-18","arxiv_id":"2207.08363","n_code_links":0,"syntology":null},{"paper":null,"slug":"pilc-practical-image-lossless-compression-1","title":"PILC: Practical Image Lossless Compression with an End-to-end GPU Oriented Neural Framework","date":"2022-06-10","arxiv_id":"2206.05279","n_code_links":0,"syntology":null},{"paper":"/paper/draft-and-revise-effective-image-generation","slug":"draft-and-revise-effective-image-generation","title":"Draft-and-Revise: Effective Image Generation with Contextual RQ-Transformer","date":"2022-06-09","arxiv_id":"2206.04452","n_code_links":0,"syntology":null},{"paper":null,"slug":"divae-photorealistic-images-synthesis-with","title":"DiVAE: Photorealistic Images Synthesis with Denoising Diffusion Decoder","date":"2022-06-01","arxiv_id":"2206.00386","n_code_links":0,"syntology":null},{"paper":"/paper/sq-vae-variational-bayes-on-discrete","slug":"sq-vae-variational-bayes-on-discrete","title":"SQ-VAE: Variational Bayes on Discrete Representation with Self-annealed Stochastic Quantization","date":"2022-05-16","arxiv_id":"2205.07547","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":1,"n_instrument":3,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sony/sqvae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-to-listen-modeling-non-deterministic","title":"Learning to Listen: Modeling Non-Deterministic Dyadic Facial Motion","date":"2022-04-18","arxiv_id":"2204.08451","n_code_links":0,"syntology":null},{"paper":"/paper/unconditional-image-text-pair-generation-with","slug":"unconditional-image-text-pair-generation-with","title":"Unconditional Image-Text Pair Generation with Multimodal Cross Quantizer","date":"2022-04-15","arxiv_id":"2204.07537","n_code_links":1,"syntology":null},{"paper":null,"slug":"analysis-of-voice-conversion-and-code","title":"Analysis of Voice Conversion and Code-Switching Synthesis Using VQ-VAE","date":"2022-03-28","arxiv_id":"2203.14640","n_code_links":0,"syntology":null},{"paper":"/paper/pixel-vq-vaes-for-improved-pixel-art","slug":"pixel-vq-vaes-for-improved-pixel-art","title":"Pixel VQ-VAEs for Improved Pixel Art Representation","date":"2022-03-23","arxiv_id":"2203.12130","n_code_links":1,"syntology":null},{"paper":"/paper/diffusion-bridges-vector-quantized","slug":"diffusion-bridges-vector-quantized","title":"Diffusion bridges vector quantized Variational AutoEncoders","date":"2022-02-10","arxiv_id":"2202.04895","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["maxjcohen/diffusion-bridges"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"robust-vector-quantized-variational","title":"Robust Vector Quantized-Variational Autoencoder","date":"2022-02-04","arxiv_id":"2202.01987","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-deep-music-generation-methods","title":"Evaluating Deep Music Generation Methods Using Data Augmentation","date":"2021-12-31","arxiv_id":"2201.00052","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-context-with-discrete-diffusion-in","title":"Global Context with Discrete Diffusion in Vector Quantised Modelling for Image Generation","date":"2021-12-03","arxiv_id":"2112.01799","n_code_links":0,"syntology":null},{"paper":"/paper/transfer-learning-with-jukebox-for-music","slug":"transfer-learning-with-jukebox-for-music","title":"Transfer Learning with Jukebox for Music Source Separation","date":"2021-11-28","arxiv_id":"2111.14200","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-model-of-semantic-completion-in-generative","title":"A model of semantic completion in generative episodic memory","date":"2021-11-26","arxiv_id":"2111.13537","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-source-aware-representations-of","title":"Learning source-aware representations of music in a discrete latent space","date":"2021-11-26","arxiv_id":"2111.13321","n_code_links":0,"syntology":null},{"paper":null,"slug":"layered-controllable-video-generation","title":"Layered Controllable Video Generation","date":"2021-11-24","arxiv_id":"2111.12747","n_code_links":0,"syntology":null},{"paper":"/paper/l-verse-bidirectional-generation-between","slug":"l-verse-bidirectional-generation-between","title":"L-Verse: Bidirectional Generation Between Image and Text","date":"2021-11-22","arxiv_id":"2111.11133","n_code_links":1,"syntology":null},{"paper":"/paper/unsupervised-source-separation-by-steering","slug":"unsupervised-source-separation-by-steering","title":"Unsupervised Source Separation By Steering Pretrained Music Models","date":"2021-10-25","arxiv_id":"2110.13071","n_code_links":1,"syntology":null},{"paper":null,"slug":"discrete-acoustic-space-for-an-efficient","title":"Discrete Acoustic Space for an Efficient Sampling in Neural Text-To-Speech","date":"2021-10-24","arxiv_id":"2110.12539","n_code_links":0,"syntology":null},{"paper":"/paper/taming-visually-guided-sound-generation","slug":"taming-visually-guided-sound-generation","title":"Taming Visually Guided Sound Generation","date":"2021-10-17","arxiv_id":"2110.08791","n_code_links":3,"syntology":{"ran":7,"of":7,"n_ran_checked":6,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["v-iashin/SpecVQGAN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"karasinger-score-free-singing-voice-synthesis","title":"KaraSinger: Score-Free Singing Voice Synthesis with VQ-VAE using Mel-spectrograms","date":"2021-10-08","arxiv_id":"2110.04005","n_code_links":0,"syntology":null},{"paper":"/paper/an-unsupervised-video-game-playstyle-metric","slug":"an-unsupervised-video-game-playstyle-metric","title":"An Unsupervised Video Game Playstyle Metric via State Discretization","date":"2021-10-03","arxiv_id":"2110.00950","n_code_links":1,"syntology":null},{"paper":null,"slug":"generating-antimicrobial-peptides-from-latent","title":"Generating Antimicrobial Peptides from Latent Secondary Structure Space","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"applying-the-information-bottleneck-principle","title":"Applying the Information Bottleneck Principle to Prosodic Representation Learning","date":"2021-08-05","arxiv_id":"2108.02821","n_code_links":0,"syntology":null},{"paper":null,"slug":"rockgpt-reconstructing-three-dimensional","title":"RockGPT: Reconstructing three-dimensional digital rocks from single two-dimensional slice from the perspective of video generation","date":"2021-08-05","arxiv_id":"2108.03132","n_code_links":0,"syntology":null},{"paper":"/paper/codified-audio-language-modeling-learns","slug":"codified-audio-language-modeling-learns","title":"Codified audio language modeling learns useful representations for music information retrieval","date":"2021-07-12","arxiv_id":"2107.05677","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["p-lambda/jukemir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vimpac-video-pre-training-via-masked-token","slug":"vimpac-video-pre-training-via-masked-token","title":"VIMPAC: Video Pre-Training via Masked Token Prediction and Contrastive Learning","date":"2021-06-21","arxiv_id":"2106.11250","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["airsplay/vimpac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/neural-distributed-source-coding","slug":"neural-distributed-source-coding","title":"Neural Distributed Source Coding","date":"2021-06-05","arxiv_id":"2106.02797","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["acnagle/neural-dsc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/factorising-meaning-and-form-for-intent","slug":"factorising-meaning-and-form-for-intent","title":"Factorising Meaning and Form for Intent-Preserving Paraphrasing","date":"2021-05-31","arxiv_id":"2105.15053","n_code_links":1,"syntology":null},{"paper":"/paper/cogview-mastering-text-to-image-generation","slug":"cogview-mastering-text-to-image-generation","title":"CogView: Mastering Text-to-Image Generation via Transformers","date":"2021-05-26","arxiv_id":"2105.13290","n_code_links":4,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["THUDM/CogView"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-disentanglement-with-multilingual","slug":"exploring-disentanglement-with-multilingual","title":"Exploring Disentanglement with Multilingual and Monolingual VQ-VAE","date":"2021-05-04","arxiv_id":"2105.01573","n_code_links":1,"syntology":null},{"paper":"/paper/videogpt-video-generation-using-vq-vae-and","slug":"videogpt-video-generation-using-vq-vae-and","title":"VideoGPT: Video Generation using VQ-VAE and Transformers","date":"2021-04-20","arxiv_id":"2104.10157","n_code_links":3,"syntology":null},{"paper":"/paper/generating-diverse-structure-for-image","slug":"generating-diverse-structure-for-image","title":"Generating Diverse Structure for Image Inpainting With Hierarchical VQ-VAE","date":"2021-03-18","arxiv_id":"2103.10022","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["USTC-JialunPeng/Diverse-Structure-Inpainting"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"wav2vec-c-a-self-supervised-model-for-speech","title":"Wav2vec-C: A Self-supervised Model for Speech Representation Learning","date":"2021-03-09","arxiv_id":"2103.08393","n_code_links":0,"syntology":null},{"paper":"/paper/generating-images-with-sparse-representations","slug":"generating-images-with-sparse-representations","title":"Generating Images with Sparse Representations","date":"2021-03-05","arxiv_id":"2103.03841","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/predicting-video-with-vqvae-1","slug":"predicting-video-with-vqvae-1","title":"Predicting Video with VQVAE","date":"2021-03-02","arxiv_id":"2103.01950","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-into-the-codec-noise-robust-speech","title":"Enhancing into the codec: Noise Robust Speech Coding with Vector-Quantized Autoencoders","date":"2021-02-12","arxiv_id":"2102.06610","n_code_links":0,"syntology":null},{"paper":"/paper/self-supervised-vq-vae-for-one-shot-music","slug":"self-supervised-vq-vae-for-one-shot-music","title":"Self-Supervised VQ-VAE for One-Shot Music Style Transfer","date":"2021-02-10","arxiv_id":"2102.05749","n_code_links":1,"syntology":null},{"paper":null,"slug":"videogen-generative-modeling-of-videos-using","title":"VideoGen: Generative Modeling of Videos using VQ-VAE and Transformers","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"anomaly-detection-through-latent-space","title":"Anomaly detection through latent space restoration using vector-quantized variational autoencoders","date":"2020-12-12","arxiv_id":"2012.06765","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-instrumentalist-net-unsupervised","title":"Multi-Instrumentalist Net: Unsupervised Generation of Music from Body Movements","date":"2020-12-07","arxiv_id":"2012.03478","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-vector-quantized-shape-code-for","title":"Learning Vector Quantized Shape Code for Amodal Blastomere Instance Segmentation","date":"2020-12-02","arxiv_id":"2012.00985","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-discrete-latent-variable","title":"A Comparison of Discrete Latent Variable Models for Speech Representation Learning","date":"2020-10-24","arxiv_id":"2010.14230","n_code_links":0,"syntology":null},{"paper":"/paper/learning-disentangled-phone-and-speaker","slug":"learning-disentangled-phone-and-speaker","title":"Learning Disentangled Phone and Speaker Representations in a Semi-Supervised VQ-VAE Paradigm","date":"2020-10-21","arxiv_id":"2010.10727","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["rhoposit/icassp2021"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"end-to-end-text-to-speech-using-latent","title":"End-to-End Text-to-Speech using Latent Duration based on VQ-VAE","date":"2020-10-19","arxiv_id":"2010.09602","n_code_links":0,"syntology":null},{"paper":"/paper/the-utility-of-decorrelating-colour-spaces-in","slug":"the-utility-of-decorrelating-colour-spaces-in","title":"The Utility of Decorrelating Colour Spaces in Vector Quantised Variational Autoencoders","date":"2020-09-30","arxiv_id":"2009.14487","n_code_links":1,"syntology":null},{"paper":null,"slug":"incorporating-reinforced-adversarial-learning","title":"Incorporating Reinforced Adversarial Learning in Autoregressive Image Generation","date":"2020-07-20","arxiv_id":"2007.09923","n_code_links":0,"syntology":null},{"paper":"/paper/generating-annotated-high-fidelity-images","slug":"generating-annotated-high-fidelity-images","title":"Generating Annotated High-Fidelity Images Containing Multiple Coherent Objects","date":"2020-06-22","arxiv_id":"2006.12150","n_code_links":1,"syntology":null},{"paper":null,"slug":"uwspeech-speech-to-speech-translation-for","title":"UWSpeech: Speech to Speech Translation for Unwritten Languages","date":"2020-06-14","arxiv_id":"2006.07926","n_code_links":0,"syntology":null},{"paper":null,"slug":"transformer-vq-vae-for-unsupervised-unit","title":"Transformer VQ-VAE for Unsupervised Unit Discovery and Speech Synthesis: ZeroSpeech 2020 Challenge","date":"2020-05-24","arxiv_id":"2005.11676","n_code_links":0,"syntology":null},{"paper":"/paper/vector-quantized-neural-networks-for-acoustic","slug":"vector-quantized-neural-networks-for-acoustic","title":"Vector-quantized neural networks for acoustic unit discovery in the ZeroSpeech 2020 challenge","date":"2020-05-19","arxiv_id":"2005.09409","n_code_links":2,"syntology":null},{"paper":"/paper/robust-training-of-vector-quantized","slug":"robust-training-of-vector-quantized","title":"Robust Training of Vector Quantized Bottleneck Models","date":"2020-05-18","arxiv_id":"2005.08520","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["distsup/DistSup"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"discretalk-text-to-speech-as-a-machine","title":"DiscreTalk: Text-to-Speech as a Machine Translation Problem","date":"2020-05-12","arxiv_id":"2005.05525","n_code_links":0,"syntology":null},{"paper":"/paper/jukebox-a-generative-model-for-music-1","slug":"jukebox-a-generative-model-for-music-1","title":"Jukebox: A Generative Model for Music","date":"2020-04-30","arxiv_id":"2005.00341","n_code_links":12,"syntology":null},{"paper":null,"slug":"adversarial-feature-learning-and-unsupervised","title":"Adversarial Feature Learning and Unsupervised Clustering based Speech Synthesis for Found Data with Acoustic and Textual Noise","date":"2020-04-28","arxiv_id":"2004.13595","n_code_links":0,"syntology":null},{"paper":"/paper/hierarchical-quantized-autoencoders","slug":"hierarchical-quantized-autoencoders","title":"Hierarchical Quantized Autoencoders","date":"2020-02-19","arxiv_id":"2002.08111","n_code_links":1,"syntology":null},{"paper":null,"slug":"neuromorphologicaly-preserving-volumetric","title":"Neuromorphologicaly-preserving Volumetric data encoding using VQ-VAE","date":"2020-02-13","arxiv_id":"2002.05692","n_code_links":0,"syntology":null},{"paper":null,"slug":"semi-supervised-grasp-detection-by","title":"Semi-supervised Grasp Detection by Representation Learning in a Vector Quantized Latent Space","date":"2020-01-23","arxiv_id":"2001.08477","n_code_links":0,"syntology":null},{"paper":null,"slug":"low-bit-rate-speech-coding-with-vq-vae-and-a","title":"Low Bit-Rate Speech Coding with VQ-VAE and a WaveNet Decoder","date":"2019-10-14","arxiv_id":"1910.06464","n_code_links":0,"syntology":null},{"paper":"/paper/melgan-generative-adversarial-networks-for","slug":"melgan-generative-adversarial-networks-for","title":"MelGAN: Generative Adversarial Networks for Conditional Waveform Synthesis","date":"2019-10-08","arxiv_id":"1910.06711","n_code_links":21,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["descriptinc/melgan-neurips"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"vqvae-unsupervised-unit-discovery-and-multi","title":"VQVAE Unsupervised Unit Discovery and Multi-scale Code2Spec Inverter for Zerospeech Challenge 2019","date":"2019-05-27","arxiv_id":"1905.11449","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-a-better-understanding-of-vector","title":"Towards a better understanding of Vector Quantized Autoencoders","date":"2019-05-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-acoustic-unit-discovery-for","title":"Unsupervised acoustic unit discovery for speech synthesis using discrete latent-variable neural networks","date":"2019-04-16","arxiv_id":"1904.07556","n_code_links":0,"syntology":null},{"paper":"/paper/unsupervised-speech-representation-learning","slug":"unsupervised-speech-representation-learning","title":"Unsupervised speech representation learning using WaveNet autoencoders","date":"2019-01-25","arxiv_id":"1901.08810","n_code_links":5,"syntology":null},{"paper":null,"slug":"variational-information-bottleneck-on-vector","title":"Variational Information Bottleneck on Vector Quantized Autoencoders","date":"2018-08-02","arxiv_id":"1808.01048","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-product-codebooks-using-vector","title":"Learning Product Codebooks using Vector Quantized Autoencoders for Image Retrieval","date":"2018-07-12","arxiv_id":"1807.04629","n_code_links":0,"syntology":null},{"paper":"/paper/neural-discrete-representation-learning","slug":"neural-discrete-representation-learning","title":"Neural Discrete Representation Learning","date":"2017-11-02","arxiv_id":"1711.00937","n_code_links":50,"syntology":{"ran":64,"of":80,"n_ran_checked":60,"n_instrument":4,"unverified":16,"pointer_only":32,"phrase":"64 ran (of which 35 constructed an object rather than computing a result; 60 with no instrument failure: 0 honoured, 0 violated, 60 with no contract checked; 4 where Syntology's instrument failed) · 16 unverified","official":{"repos":["deepmind/sonnet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"e67297a3665e0fe938a46e54450c0916c8bbde01917503af217c9c80ae460b32","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}