{"url":"/method/rem","slug":"rem","name":"REM","full_name":"Random Ensemble Mixture","full_name_withheld":false,"description_markdown":"Random Ensemble Mixture (REM) is an easy to implement extension of [DQN](https://paperswithcode.com/method/dqn) inspired by [Dropout](https://paperswithcode.com/method/dropout). The key intuition behind REM is that if one has access to multiple estimates of Q-values, then a weighted combination of the Q-value estimates is also an estimate for Q-values. Accordingly, in each training step, REM randomly combines multiple Q-value estimates and uses this random combination for robust training.","description_state":"present","introduced_year":null,"introduced_by":{"title":"An Optimistic Perspective on Offline Reinforcement Learning","paper":"/paper/striving-for-simplicity-in-off-policy-deep","first_author":"Rishabh Agarwal","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/striving-for-simplicity-in-off-policy-deep"},"source":{"url":"https://arxiv.org/abs/1907.04543v4","title":"An Optimistic Perspective on Offline Reinforcement Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Randomized Value Functions","url":"/methods/category/randomized-value-functions","pwc_aliases":[]},{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Off-Policy TD Control","url":"/methods/category/off-policy-td-control","pwc_aliases":[]},{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Q-Learning Networks","url":"/methods/category/q-learning-networks","pwc_aliases":["q-learning"]}],"n_papers_tagged":48,"archive_num_papers":48,"papers_newest_first":[{"paper":null,"title":"Getting More from Less: Transfer Learning Improves Sleep Stage Decoding Accuracy in Peripheral Wearable Devices","date":"2025-05-31","arxiv_id":"2506.00730","n_code_links":0,"syntology":null},{"paper":null,"title":"REM: A Scalable Reinforced Multi-Expert Framework for Multiplex Influence Maximization","date":"2025-01-01","arxiv_id":"2501.00779","n_code_links":0,"syntology":null},{"paper":"/paper/mobilenetv2-a-lightweight-classification","title":"MobileNetV2: A lightweight classification model for home-based sleep apnea screening","date":"2024-12-28","arxiv_id":"2412.19967","n_code_links":1,"syntology":null},{"paper":null,"title":"ECG-SleepNet: Deep Learning-Based Comprehensive Sleep Stage Classification Using ECG Signals","date":"2024-12-02","arxiv_id":"2412.01929","n_code_links":0,"syntology":null},{"paper":null,"title":"ReferEverything: Towards Segmenting Everything We Can Speak of in Videos","date":"2024-10-30","arxiv_id":"2410.23287","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-and-private-marginal-reconstruction","title":"Efficient and Private Marginal Reconstruction with Local Non-Negativity","date":"2024-10-01","arxiv_id":"2410.01091","n_code_links":1,"syntology":{"ran":17,"of":25,"unverified":8,"pointer_only":25}},{"paper":null,"title":"Optimizing Photoplethysmography-Based Sleep Staging Models by Leveraging Temporal Context for Wearable Devices Applications","date":"2024-10-01","arxiv_id":"2410.00693","n_code_links":0,"syntology":null},{"paper":"/paper/post-hoc-utterance-refining-method-by-entity","title":"Post-hoc Utterance Refining Method by Entity Mining for Faithful Knowledge Grounded Conversations","date":"2024-06-16","arxiv_id":"2406.10809","n_code_links":1,"syntology":null},{"paper":null,"title":"Evaluating the Influence of Temporal Context on Automatic Mouse Sleep Staging through the Application of Human Models","date":"2024-06-06","arxiv_id":"2406.16911","n_code_links":0,"syntology":null},{"paper":null,"title":"Nonlinear Transformations Against Unlearnable Datasets","date":"2024-06-05","arxiv_id":"2406.02883","n_code_links":0,"syntology":null},{"paper":null,"title":"Spatial-temporal analysis of neural desynchronization in sleep-like states reveals critical dynamics","date":"2024-05-28","arxiv_id":"2405.18329","n_code_links":0,"syntology":null},{"paper":null,"title":"Sparse Bayesian Learning-Based Hierarchical Construction for 3D Radio Environment Maps Incorporating Channel Shadowing","date":"2024-03-13","arxiv_id":"2403.08323","n_code_links":0,"syntology":null},{"paper":null,"title":"Interference Mitigation in LEO Constellations with Limited Radio Environment Information","date":"2024-02-19","arxiv_id":"2402.12103","n_code_links":0,"syntology":null},{"paper":"/paper/improving-token-based-world-models-with","title":"Improving Token-Based World Models with Parallel Observation Prediction","date":"2024-02-08","arxiv_id":"2402.05643","n_code_links":1,"syntology":{"ran":7,"of":9,"unverified":2,"pointer_only":9}},{"paper":"/paper/attention-based-cnn-bilstm-for-sleep-state","title":"Attention-Based CNN-BiLSTM for Sleep State Classification of Spatiotemporal Wide-Field Calcium Imaging Data","date":"2024-01-16","arxiv_id":"2401.08098","n_code_links":1,"syntology":null},{"paper":null,"title":"Inverse-like Antagonistic Scene Text Spotting via Reading-Order Estimation and Dynamic Sampling","date":"2024-01-08","arxiv_id":"2401.03637","n_code_links":0,"syntology":null},{"paper":"/paper/modeling-non-linear-effects-with-neural","title":"Modeling non-linear Effects with Neural Networks in Relational Event Models","date":"2023-12-19","arxiv_id":"2312.12357","n_code_links":1,"syntology":null},{"paper":null,"title":"Wake-Sleep Consolidated Learning","date":"2023-12-06","arxiv_id":"2401.08623","n_code_links":0,"syntology":null},{"paper":null,"title":"Two-compartment neuronal spiking model expressing brain-state specific apical-amplification, -isolation and -drive regimes","date":"2023-11-10","arxiv_id":"2311.06074","n_code_links":0,"syntology":null},{"paper":null,"title":"SI-SD: Sleep Interpreter through awake-guided cross-subject Semantic Decoding","date":"2023-09-28","arxiv_id":"2309.16457","n_code_links":0,"syntology":null},{"paper":null,"title":"Enhancing Healthcare with EOG: A Novel Approach to Sleep Stage Classification","date":"2023-09-25","arxiv_id":"2310.03757","n_code_links":0,"syntology":null},{"paper":null,"title":"Coalescent processes emerging from large deviations","date":"2023-08-28","arxiv_id":"2308.14715","n_code_links":0,"syntology":null},{"paper":null,"title":"Goal-Conditioned Reinforcement Learning with Disentanglement-based Reachability Planning","date":"2023-07-20","arxiv_id":"2307.10846","n_code_links":0,"syntology":null},{"paper":null,"title":"How does agency impact human-AI collaborative design space exploration? A case study on ship design with deep generative models","date":"2023-05-16","arxiv_id":"2305.10451","n_code_links":0,"syntology":null},{"paper":null,"title":"Position Bias Estimation with Item Embedding for Sparse Dataset","date":"2023-05-10","arxiv_id":"2305.13931","n_code_links":0,"syntology":null},{"paper":null,"title":"Computational role of sleep in memory reorganization","date":"2023-04-06","arxiv_id":"2304.02873","n_code_links":0,"syntology":null},{"paper":null,"title":"Recovering Arrhythmic EEG Transients from Their Stochastic Interference","date":"2023-03-14","arxiv_id":"2303.07683","n_code_links":0,"syntology":null},{"paper":null,"title":"NREM and REM: cognitive and energetic gains in thalamo-cortical sleeping and awake spiking model","date":"2022-11-13","arxiv_id":"2211.06889","n_code_links":0,"syntology":null},{"paper":null,"title":"Continual learning benefits from multiple sleep mechanisms: NREM, REM, and Synaptic Downscaling","date":"2022-09-09","arxiv_id":"2209.05245","n_code_links":0,"syntology":null},{"paper":"/paper/disentangled-modeling-of-domain-and-relevance","title":"Disentangled Modeling of Domain and Relevance for Adaptable Dense Retrieval","date":"2022-08-11","arxiv_id":"2208.05753","n_code_links":1,"syntology":null}],"papers_shown":30,"tasks":[{"task":"/task/eeg-1","name":"EEG","papers":6},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":6},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":5},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":5},{"task":"/task/dqn-replay-dataset","name":"DQN Replay Dataset","papers":3},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":3},{"task":"/task/offline-rl","name":"Offline RL","papers":3},{"task":"/task/atari-games","name":"Atari Games","papers":2},{"task":"/task/classification-1","name":"Classification","papers":2},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":2},{"task":"/task/continual-learning","name":"Continual Learning","papers":2},{"task":"/task/diversity","name":"Diversity","papers":2},{"task":"/task/management","name":"Management","papers":2},{"task":"/task/object","name":"Object","papers":2},{"task":"/task/q-learning","name":"Q-Learning","papers":2},{"task":null,"name":"Relation","papers":2},{"task":"/task/sleep-staging","name":"Sleep Staging","papers":2},{"task":"/task/super-resolution","name":"Super-Resolution","papers":2},{"task":"/task/ad-hoc-information-retrieval","name":"Ad-Hoc Information Retrieval","papers":1},{"task":"/task/additive-models","name":"Additive models","papers":1}],"tasks_shown":20,"n_tasks":69,"usage_by_year":[{"year":"2019","papers":1},{"year":"2020","papers":4},{"year":"2021","papers":11},{"year":"2022","papers":5},{"year":"2023","papers":11},{"year":"2024","papers":14},{"year":"2025","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/rem"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}