{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/decoder/papers/58","list_of":"/task/decoder","task":"Decoder","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":58,"pages_in_order":104,"rows_per_page":100,"rows":[5701,5800],"of":10368,"counts":{"archive_papers_tagged":10368,"with_a_code_link":4358,"where_syntology_ran_a_sample":1061,"not_listed_spam_title":0,"listed":10368,"listed_where_code_ran":1061,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":909,"every_run_a_failure_of_syntologys_instrument":152,"listed_with_a_run_with_no_instrument_failure":909,"listed_every_run_a_failure_of_syntologys_instrument":152,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/decoder","prev":"/task/decoder/papers/57","next":"/task/decoder/papers/59","papers":[{"url":null,"slug":"robust-policy-learning-via-offline-skill","title":"Robust Policy Learning via Offline Skill Diffusion","date":"2024-03-01","arxiv_id":"2403.00225","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-generative-ai-for-portuguese-with","title":"Advancing Generative AI for Portuguese with Open Decoder Gervásio PT*","date":"2024-02-29","arxiv_id":"2402.18766","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-speech-translation-models-via","title":"Compact Speech Translation Models via Discrete Speech Units Pretraining","date":"2024-02-29","arxiv_id":"2402.19333","repositories_listed":0,"syntology":null},{"url":null,"slug":"sne-roadsegv2-advancing-heterogeneous-feature","title":"SNE-RoadSegV2: Advancing Heterogeneous Feature Fusion and Fallibility Awareness for Freespace Detection","date":"2024-02-29","arxiv_id":"2402.18918","repositories_listed":0,"syntology":null},{"url":null,"slug":"ean-mapnet-efficient-vectorized-hd-map","title":"EAN-MapNet: Efficient Vectorized HD Map Construction with Anchor Neighborhoods","date":"2024-02-28","arxiv_id":"2402.18278","repositories_listed":0,"syntology":null},{"url":null,"slug":"graphpub-generation-of-differential-privacy","title":"GraphPub: Generation of Differential Privacy Graph with High Availability","date":"2024-02-28","arxiv_id":"2403.00030","repositories_listed":0,"syntology":null},{"url":null,"slug":"nerv-an-enhanced-implicit-neural-video","title":"NERV++: An Enhanced Implicit Neural Video Representation","date":"2024-02-28","arxiv_id":"2402.18305","repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-encoder-output-frame-rate-reduction","title":"Extreme Encoder Output Frame Rate Reduction: Improving Computational Latencies of Large End-to-End Models","date":"2024-02-27","arxiv_id":"2402.17184","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-visual-speech-enhancement-in-noisy","title":"Audio-Visual Speech Enhancement in Noisy Environments via Emotion-Based Contextual Cues","date":"2024-02-26","arxiv_id":"2402.16394","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-bit-distortion-free-watermarking-for","title":"Multi-Bit Distortion-Free Watermarking for Large Language Models","date":"2024-02-26","arxiv_id":"2402.16578","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallelized-spatiotemporal-binding","title":"Parallelized Spatiotemporal Binding","date":"2024-02-26","arxiv_id":"2402.17077","repositories_listed":0,"syntology":null},{"url":null,"slug":"think-big-generate-quick-llm-to-slm-for-fast","title":"Think Big, Generate Quick: LLM-to-SLM for Fast Autoregressive Decoding","date":"2024-02-26","arxiv_id":"2402.16844","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-to-essay-generation-with-knowledge","title":"Topic-to-essay generation with knowledge-based content selection","date":"2024-02-26","arxiv_id":"2402.16248","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-flexible-machine-learning-models-for","title":"Building Flexible Machine Learning Models for Scientific Computing at Scale","date":"2024-02-25","arxiv_id":"2402.16014","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-based-user-profiles-for","title":"Language-Based User Profiles for Recommendation","date":"2024-02-23","arxiv_id":"2402.15623","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-surprising-effectiveness-of-skip-tuning","title":"The Surprising Effectiveness of Skip-Tuning in Diffusion Sampling","date":"2024-02-23","arxiv_id":"2402.15170","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-self-supervised-pressure-map-human-keypoint","title":"A Self-supervised Pressure Map human keypoint Detection Approch: Optimizing Generalization and Computational Efficiency Across Datasets","date":"2024-02-22","arxiv_id":"2402.14241","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-radiance-fields-for-edge-video","title":"Distributed Radiance Fields for Edge Video Compression and Metaverse Integration in Autonomous Driving","date":"2024-02-22","arxiv_id":"2402.14642","repositories_listed":0,"syntology":null},{"url":null,"slug":"polynet-learning-diverse-solution-strategies","title":"PolyNet: Learning Diverse Solution Strategies for Neural Combinatorial Optimization","date":"2024-02-21","arxiv_id":"2402.14048","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-thought-empowers-transformers-to","title":"Chain of Thought Empowers Transformers to Solve Inherently Serial Problems","date":"2024-02-20","arxiv_id":"2402.12875","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparison-of-conventional-hybrid-and-ctc","title":"Comparison of Conventional Hybrid and CTC/Attention Decoders for Continuous Visual Speech Recognition","date":"2024-02-20","arxiv_id":"2402.13004","repositories_listed":0,"syntology":null},{"url":null,"slug":"gloria-a-generative-and-open-large-language","title":"GlórIA - A Generative and Open Large Language Model for Portuguese","date":"2024-02-20","arxiv_id":"2402.12969","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-domain-invariant-temporal-dynamics","title":"Learning Causal Domain-Invariant Temporal Dynamics for Few-Shot Action Recognition","date":"2024-02-20","arxiv_id":"2402.12706","repositories_listed":0,"syntology":null},{"url":null,"slug":"linksage-optimizing-job-matching-using-graph","title":"LinkSAGE: Optimizing Job Matching Using Graph Neural Networks","date":"2024-02-20","arxiv_id":"2402.13430","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-transformers-revolutionizing-the","title":"Toward TransfORmers: Revolutionizing the Solution of Mixed Integer Programs with Transformers","date":"2024-02-20","arxiv_id":"2402.13380","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformer-based-learned-image-compression","title":"Transformer-based Learned Image Compression for Joint Decoding and Denoising","date":"2024-02-20","arxiv_id":"2402.12888","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-levenshtein-transformer-s-decoder","title":"Analysis of Levenshtein Transformer's Decoder and Its Variants","date":"2024-02-19","arxiv_id":"2402.12249","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-based-prediction-of-ditching","title":"Machine Learning based Prediction of Ditching Loads","date":"2024-02-16","arxiv_id":"2402.10724","repositories_listed":0,"syntology":null},{"url":null,"slug":"camouflage-is-all-you-need-evaluating-and","title":"Camouflage is all you need: Evaluating and Enhancing Language Model Robustness Against Camouflage Adversarial Attacks","date":"2024-02-15","arxiv_id":"2402.09874","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-language-adaptive-pre-training","title":"Efficient Language Adaptive Pre-training: Extending State-of-the-Art Large Language Models for Polish","date":"2024-02-15","arxiv_id":"2402.09759","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-of-pretrained-language-models-on","title":"Knowledge of Pretrained Language Models on Surface Information of Tokens","date":"2024-02-15","arxiv_id":"2402.09808","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-the-energy-demand-of-a-hardware","title":"Energy Demand Prediction for Hardware Video Decoders Using Software Profiling","date":"2024-02-15","arxiv_id":"2402.09926","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-review-of-software-and","title":"A Comprehensive Review of Software and Hardware Energy Efficiency of Video Decoders","date":"2024-02-14","arxiv_id":"2402.09001","repositories_listed":0,"syntology":null},{"url":"/paper/mobilespeech-a-fast-and-high-fidelity","slug":"mobilespeech-a-fast-and-high-fidelity","title":"MobileSpeech: A Fast and High-Fidelity Framework for Mobile Zero-Shot Text-to-Speech","date":"2024-02-14","arxiv_id":"2402.09378","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":6,"n_ran_checked":7,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mobilespeech-a-fast-and-high-fidelity#ran","syntology_url":"https://syntology.ai/paper/2402.09378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09378"}},"official":null}},{"url":null,"slug":"moving-object-proposals-with-deep-learned","title":"Moving Object Proposals with Deep Learned Optical Flow for Video Object Segmentation","date":"2024-02-14","arxiv_id":"2402.08882","repositories_listed":0,"syntology":null},{"url":null,"slug":"unienc-cassnat-an-encoder-only-non","title":"UniEnc-CASSNAT: An Encoder-only Non-autoregressive ASR for Speech SSL Models","date":"2024-02-14","arxiv_id":"2402.08898","repositories_listed":0,"syntology":null},{"url":null,"slug":"base-tts-lessons-from-building-a-billion","title":"BASE TTS: Lessons from building a billion-parameter Text-to-Speech model on 100K hours of data","date":"2024-02-12","arxiv_id":"2402.08093","repositories_listed":0,"syntology":null},{"url":null,"slug":"american-sign-language-video-to-text","title":"American Sign Language Video to Text Translation","date":"2024-02-11","arxiv_id":"2402.07255","repositories_listed":0,"syntology":null},{"url":null,"slug":"sportsngen-sustained-generation-of-realistic","title":"SportsNGEN: Sustained Generation of Realistic Multi-player Sports Gameplay","date":"2024-02-10","arxiv_id":"2403.12977","repositories_listed":0,"syntology":null},{"url":null,"slug":"curveformer-3d-lane-detection-by-curve-1","title":"CurveFormer++: 3D Lane Detection by Curve Propagation with Temporal Curve Queries and Attention","date":"2024-02-09","arxiv_id":"2402.06423","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-captioning-and","title":"Large Language Models for Captioning and Retrieving Remote Sensing Images","date":"2024-02-09","arxiv_id":"2402.06475","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-2d-neural-nets-for-phase-retrieval-in","title":"3D-2D Neural Nets for Phase Retrieval in Noisy Interferometric Imaging","date":"2024-02-08","arxiv_id":"2402.06063","repositories_listed":0,"syntology":null},{"url":null,"slug":"capability-enhancement-of-the-x-ray-micro","title":"Capability enhancement of the X-ray micro-tomography system via ML-assisted approaches","date":"2024-02-08","arxiv_id":"2402.05983","repositories_listed":0,"syntology":null},{"url":null,"slug":"geneft-understanding-statics-and-dynamics-of","title":"GenEFT: Understanding Statics and Dynamics of Model Generalization via Effective Theory","date":"2024-02-08","arxiv_id":"2402.05916","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-bias-and-fast-convergence-rates-for","title":"Implicit Bias and Fast Convergence Rates for Self-attention","date":"2024-02-08","arxiv_id":"2402.05738","repositories_listed":0,"syntology":null},{"url":null,"slug":"spirdet-towards-efficient-accurate-and","title":"SpirDet: Towards Efficient, Accurate and Lightweight Infrared Small Target Detector","date":"2024-02-08","arxiv_id":"2402.05410","repositories_listed":0,"syntology":null},{"url":null,"slug":"compression-of-structured-data-with","title":"Compression of Structured Data with Autoencoders: Provable Benefit of Nonlinearities and Depth","date":"2024-02-07","arxiv_id":"2402.05013","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-bridge-generative-model-based-domain","title":"Domain Bridge: Generative model-based domain forensic for black-box models","date":"2024-02-07","arxiv_id":"2402.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"stablemask-refining-causal-masking-in-decoder","title":"StableMask: Refining Causal Masking in Decoder-only Transformer","date":"2024-02-07","arxiv_id":"2402.04779","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-encoder-optimized-pam-im-dd-transceivers","title":"Auto-Encoder Optimized PAM IM/DD Transceivers for Amplified Fiber Links","date":"2024-02-06","arxiv_id":"2402.04395","repositories_listed":0,"syntology":null},{"url":null,"slug":"lens-a-foundation-model-for-network-traffic","title":"Lens: A Foundation Model for Network Traffic","date":"2024-02-06","arxiv_id":"2402.03646","repositories_listed":0,"syntology":null},{"url":null,"slug":"densely-decoded-networks-with-adaptive-deep","title":"Densely Decoded Networks with Adaptive Deep Supervision for Medical Image Segmentation","date":"2024-02-05","arxiv_id":"2402.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-the-stability-of-llm-based-speech","title":"Enhancing the Stability of LLM-based Speech Generation Systems through Self-Supervised Representations","date":"2024-02-05","arxiv_id":"2402.03407","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptual-learned-image-compression-via-end","title":"Perceptual Learned Image Compression via End-to-End JND-Based Optimization","date":"2024-02-05","arxiv_id":"2402.02836","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-audio-visual-segmentation-by","title":"Bootstrapping Audio-Visual Segmentation by Strengthening Audio Cues","date":"2024-02-04","arxiv_id":"2402.02327","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-intrinsic-properties-of-medical","title":"Exploring Intrinsic Properties of Medical Images for Self-Supervised Binary Semantic Segmentation","date":"2024-02-04","arxiv_id":"2402.02367","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-paper-the-landscape-and-challenges","title":"The Landscape and Challenges of HPC Research and LLMs","date":"2024-02-03","arxiv_id":"2402.02018","repositories_listed":0,"syntology":null},{"url":null,"slug":"recnet-an-invertible-point-cloud-encoding","title":"RecNet: An Invertible Point Cloud Encoding through Range Image Embeddings for Multi-Robot Map Sharing and Reconstruction","date":"2024-02-03","arxiv_id":"2402.02192","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-network-representations-with","title":"Learning Network Representations with Disentangled Graph Auto-Encoder","date":"2024-02-02","arxiv_id":"2402.01143","repositories_listed":0,"syntology":null},{"url":null,"slug":"spiking-centernet-a-distillation-boosted","title":"Spiking CenterNet: A Distillation-boosted Spiking Neural Network for Object Detection","date":"2024-02-02","arxiv_id":"2402.01287","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-limits-of-decoder-only-models","title":"Exploring the limits of decoder-only models trained on public speech recognition corpora","date":"2024-01-31","arxiv_id":"2402.00235","repositories_listed":0,"syntology":null},{"url":"/paper/robustly-overfitting-latents-for-flexible","slug":"robustly-overfitting-latents-for-flexible","title":"Robustly overfitting latents for flexible neural image compression","date":"2024-01-31","arxiv_id":"2401.17789","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robustly-overfitting-latents-for-flexible#ran","syntology_url":"https://syntology.ai/paper/2401.17789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17789"}},"official":null}},{"url":null,"slug":"speechcomposer-unifying-multiple-speech-tasks","title":"SpeechComposer: Unifying Multiple Speech Tasks with Prompt Composition","date":"2024-01-31","arxiv_id":"2401.18045","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-gamma-band-responses-to-the-speech","title":"Detecting gamma-band responses to the speech envelope for the ICASSP 2024 Auditory EEG Decoding Signal Processing Grand Challenge","date":"2024-01-30","arxiv_id":"2401.17380","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-for-text-diffusion-models","title":"Transfer Learning for Text Diffusion Models","date":"2024-01-30","arxiv_id":"2401.17181","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-need-one-step-fast-super-resolution","title":"You Only Need One Step: Fast Super-Resolution with Stable Diffusion via Scale Distillation","date":"2024-01-30","arxiv_id":"2401.17258","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-generative-and-discriminative-models","title":"Bridging Generative and Discriminative Models for Unified Visual Perception with Diffusion Priors","date":"2024-01-29","arxiv_id":"2401.16459","repositories_listed":0,"syntology":null},{"url":null,"slug":"drbert-unveiling-the-potential-of-masked","title":"BPDec: Unveiling the Potential of Masked Language Modeling Decoder in BERT pretraining","date":"2024-01-29","arxiv_id":"2401.15861","repositories_listed":0,"syntology":null},{"url":null,"slug":"fimp-future-interaction-modeling-for-multi","title":"FIMP: Future Interaction Modeling for Multi-Agent Motion Prediction","date":"2024-01-29","arxiv_id":"2401.16189","repositories_listed":0,"syntology":null},{"url":null,"slug":"gland-segmentation-via-dual-encoders-and","title":"Gland Segmentation Via Dual Encoders and Boundary-Enhanced Attention","date":"2024-01-29","arxiv_id":"2401.15990","repositories_listed":0,"syntology":null},{"url":null,"slug":"mv2mae-multi-view-video-masked-autoencoders","title":"MV2MAE: Multi-View Video Masked Autoencoders","date":"2024-01-29","arxiv_id":"2401.15900","repositories_listed":0,"syntology":null},{"url":null,"slug":"proto-mpc-an-encoder-prototype-decoder","title":"Proto-MPC: An Encoder-Prototype-Decoder Approach for Quadrotor Control in Challenging Winds","date":"2024-01-27","arxiv_id":"2401.15508","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-joint-source-channel-coding-for-1","title":"Deep Joint Source-Channel Coding for Efficient and Reliable Cross-Technology Communication","date":"2024-01-26","arxiv_id":"2402.10072","repositories_listed":0,"syntology":null},{"url":null,"slug":"unit-dsr-dysarthric-speech-reconstruction","title":"UNIT-DSR: Dysarthric Speech Reconstruction System Using Speech Unit Normalization","date":"2024-01-26","arxiv_id":"2401.14664","repositories_listed":0,"syntology":null},{"url":null,"slug":"vn-net-vision-numerical-fusion-graph","title":"VN-Net: Vision-Numerical Fusion Graph Convolutional Network for Sparse Spatio-Temporal Meteorological Forecasting","date":"2024-01-26","arxiv_id":"2404.16037","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-not-always-look-right-investigating-the","title":"Looking Right is Sometimes Right: Investigating the Capabilities of Decoder-only LLMs for Sequence Labeling","date":"2024-01-25","arxiv_id":"2401.14556","repositories_listed":0,"syntology":null},{"url":null,"slug":"friendly-attacks-to-improve-channel-coding","title":"Friendly Attacks to Improve Channel Coding Reliability","date":"2024-01-25","arxiv_id":"2401.14184","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-graph-supported-benchmark-and-video","title":"Knowledge Guided Entity-aware Video Captioning and A Basketball Benchmark","date":"2024-01-25","arxiv_id":"2401.13888","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-t-decoder-only-generative-transducer-for","title":"VALL-T: Decoder-Only Generative Transducer for Robust and Decoding-Controllable Text-to-Speech","date":"2024-01-25","arxiv_id":"2401.14321","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-intrusive-speech-intelligibility-1","title":"Non-Intrusive Speech Intelligibility Prediction for Hearing-Impaired Users using Intermediate ASR Features and Human Memory Models","date":"2024-01-24","arxiv_id":"2401.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptually-motivated-spatial-audio-codec","title":"Perceptually-motivated Spatial Audio Codec for Higher-Order Ambisonics Compression","date":"2024-01-24","arxiv_id":"2401.13401","repositories_listed":0,"syntology":null},{"url":null,"slug":"spactor-t5-pre-training-t5-models-with-span","title":"SpacTor-T5: Pre-training T5 Models with Span Corruption and Replaced Token Detection","date":"2024-01-24","arxiv_id":"2401.13160","repositories_listed":0,"syntology":null},{"url":"/paper/boosting-unknown-number-speaker-separation","slug":"boosting-unknown-number-speaker-separation","title":"Boosting Unknown-number Speaker Separation with Transformer Decoder-based Attractor","date":"2024-01-23","arxiv_id":"2401.12473","repositories_listed":0,"syntology":null},{"url":null,"slug":"raw-a-robust-and-agile-plug-and-play","title":"RAW: A Robust and Agile Plug-and-Play Watermark Framework for AI-Generated Images with Provable Guarantees","date":"2024-01-23","arxiv_id":"2403.18774","repositories_listed":0,"syntology":null},{"url":null,"slug":"concealed-object-segmentation-with","title":"Concealed Object Segmentation with Hierarchical Coherence Modeling","date":"2024-01-22","arxiv_id":"2401.11767","repositories_listed":0,"syntology":null},{"url":null,"slug":"keep-decoding-parallel-with-effective","title":"Keep Decoding Parallel with Effective Knowledge Distillation from Language Models to End-to-end Speech Recognisers","date":"2024-01-22","arxiv_id":"2401.11700","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2-clip-a-multimodal-multi-task-adapting","title":"M2-CLIP: A Multimodal, Multi-task Adapting Framework for Video Action Recognition","date":"2024-01-22","arxiv_id":"2401.11649","repositories_listed":0,"syntology":null},{"url":null,"slug":"rate-distortion-perception-tradeoff-based-on","title":"Rate-Distortion-Perception Tradeoff Based on the Conditional-Distribution Perception Measure","date":"2024-01-22","arxiv_id":"2401.12207","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-view-3d-human-digitalization-with","title":"Template-Free Single-View 3D Human Digitalization with Diffusion-Guided LRM","date":"2024-01-22","arxiv_id":"2401.12175","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-large-language-model-for-end-to-end","title":"Using Large Language Model for End-to-End Chinese ASR and NER","date":"2024-01-21","arxiv_id":"2401.11382","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-image-coding-for","title":"Bridging the gap between image coding for machines and humans","date":"2024-01-19","arxiv_id":"2401.10732","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-semantic-compression-for-cnn","title":"Dynamic Semantic Compression for CNN Inference in Multi-access Edge Computing: A Graph Reinforcement Learning-based Autoencoder","date":"2024-01-19","arxiv_id":"2401.12167","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-position-aware-implicit-neural","title":"Learning Position-Aware Implicit Neural Network for Real-World Face Inpainting","date":"2024-01-19","arxiv_id":"2401.10537","repositories_listed":0,"syntology":null},{"url":null,"slug":"polytopic-autoencoders-with-smooth-clustering","title":"Polytopic Autoencoders with Smooth Clustering for Reduced-order Modelling of Flows","date":"2024-01-19","arxiv_id":"2401.10620","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-generative-modeling-for-financial-time","title":"Deep Generative Modeling for Financial Time Series with Application in VaR: A Comparative Review","date":"2024-01-18","arxiv_id":"2401.10370","repositories_listed":0,"syntology":null},{"url":null,"slug":"matscire-leveraging-pointer-networks-to","title":"MatSciRE: Leveraging Pointer Networks to Automate Entity and Relation Extraction for Material Science Knowledge-base Construction","date":"2024-01-18","arxiv_id":"2401.09839","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-motion-retargeting-for-human","title":"Unsupervised Motion Retargeting for Human-Robot Imitation","date":"2024-01-18","arxiv_id":"2402.05115","repositories_listed":0,"syntology":null},{"url":null,"slug":"cfasl-composite-factor-aligned-symmetry","title":"CFASL: Composite Factor-Aligned Symmetry Learning for Disentanglement in Variational AutoEncoder","date":"2024-01-17","arxiv_id":"2401.08897","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-relation-transformer-for-contextual","title":"Dynamic Relation Transformer for Contextual Text Block Detection","date":"2024-01-17","arxiv_id":"2401.09232","repositories_listed":0,"syntology":null},{"url":null,"slug":"subwavelength-imaging-using-a-solid-immersion","title":"Subwavelength Imaging using a Solid-Immersion Diffractive Optical Processor","date":"2024-01-17","arxiv_id":"2401.08923","repositories_listed":0,"syntology":null}],"record_sha256":"a3148b0a5bcfa916e23449ab189ff9651226f774925b884c3fdc25b02d270812","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}