{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/robust-deep-reinforcement-learning-via-multi","title":"DRIBO: Robust Deep Reinforcement Learning via Multi-View Information Bottleneck","arxiv_id":"2102.13268","date":"2021-02-26","proceeding":null,"authors":["Jiameng Fan","Wenchao Li"],"abstract":"Deep reinforcement learning (DRL) agents are often sensitive to visual changes that were unseen in their training environments. To address this problem, we leverage the sequential nature of RL to learn robust representations that encode only task-relevant information from observations based on the unsupervised multi-view setting. Specifically, we introduce a novel contrastive version of the Multi-View Information Bottleneck (MIB) objective for temporal data. We train RL agents from pixels with this auxiliary objective to learn robust representations that can compress away task-irrelevant information and are predictive of task-relevant dynamics. This approach enables us to train high-performance policies that are robust to visual distractions and can generalize well to unseen environments. We demonstrate that our approach can achieve SOTA performance on a diverse set of visual control tasks in the DeepMind Control Suite when the background is replaced with natural videos. In addition, we show that our approach outperforms well-established baselines for generalization to unseen environments on the Procgen benchmark. Our code is open-sourced and available at https://github. com/BU-DEPEND-Lab/DRIBO.","url_abs":"https://arxiv.org/abs/2102.13268v4","url_pdf":"https://arxiv.org/pdf/2102.13268v4.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"robust-deep-reinforcement-learning-via-multi","repo_url":"https://github.com/BU-DEPEND-Lab/DRIBO","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"deep-reinforcement-learning","task_name":"Deep Reinforcement Learning"},{"task_slug":"reinforcement-learning","task_name":"Reinforcement Learning"},{"task_slug":"reinforcement-learning-1","task_name":"Reinforcement Learning (RL)"},{"task_slug":"representation-learning","task_name":"Representation Learning"},{"task_slug":"reinforcement-learning-2","task_name":"reinforcement-learning"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2102.13268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.13268"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/BU-DEPEND-Lab/DRIBO","reach":null}],"summary":{"ran_draft_wrong":6,"unverified":1},"by_repo_kind":{"official":{"samples":7,"ran":6,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"c86249c3d9ba943f","entry":"flatten_states","repo":"BU-DEPEND-Lab/DRIBO","repo_kind":"official","path":"DRIBO/DRIBO_sac.py","file_url":"https://github.com/BU-DEPEND-Lab/DRIBO/blob/HEAD/DRIBO/DRIBO_sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"c86249c3d9ba943f"}},{"code_sha256_prefix":"daa7b3a0355eb7c3","entry":"gaussian_logprob","repo":"bu-depend-lab/dribo","repo_kind":"official","path":"DRIBO/DRIBO_sac.py","file_url":"https://github.com/bu-depend-lab/dribo/blob/HEAD/DRIBO/DRIBO_sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"daa7b3a0355eb7c3"}},{"code_sha256_prefix":"6e2c1f59f1d53fcc","entry":"get_dist","repo":"BU-DEPEND-Lab/DRIBO","repo_kind":"official","path":"DRIBO/DRIBO_sac.py","file_url":"https://github.com/BU-DEPEND-Lab/DRIBO/blob/HEAD/DRIBO/DRIBO_sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"6e2c1f59f1d53fcc"}},{"code_sha256_prefix":"ead01b9035a66905","entry":"namedarraytuple","repo":"BU-DEPEND-Lab/DRIBO","repo_kind":"official","path":"DRIBO/DRIBO_sac.py","file_url":"https://github.com/BU-DEPEND-Lab/DRIBO/blob/HEAD/DRIBO/DRIBO_sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"ead01b9035a66905"}},{"code_sha256_prefix":"e4f442d1770eacd1","entry":"squash","repo":"bu-depend-lab/dribo","repo_kind":"official","path":"DRIBO/DRIBO_sac.py","file_url":"https://github.com/bu-depend-lab/dribo/blob/HEAD/DRIBO/DRIBO_sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"e4f442d1770eacd1"}},{"code_sha256_prefix":"0fe5032c8a208099","entry":"tuple_itemgetter","repo":"BU-DEPEND-Lab/DRIBO","repo_kind":"official","path":"DRIBO/DRIBO_sac.py","file_url":"https://github.com/BU-DEPEND-Lab/DRIBO/blob/HEAD/DRIBO/DRIBO_sac.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"0fe5032c8a208099"}},{"code_sha256_prefix":"54e2e388a1b71f60","entry":"DRIBO","repo":"BU-DEPEND-Lab/DRIBO","repo_kind":"official","path":"DRIBO/DRIBO_sac.py","file_url":"https://github.com/BU-DEPEND-Lab/DRIBO/blob/HEAD/DRIBO/DRIBO_sac.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"54e2e388a1b71f60"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}