{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/9","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":9,"pages_in_order":12,"rows_per_page":100,"rows":[801,900],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/8","next":"/method/entropy-regularization/papers/10","papers":[{"paper":null,"slug":"efficient-out-of-distribution-detection-using","title":"Efficient Out-of-Distribution Detection Using Latent Space of $β$-VAE for Cyber-Physical Systems","date":"2021-08-26","arxiv_id":"2108.11800","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-laplacian-diffusion-localization-of","title":"Graph Laplacian Diffusion Localization of Connected and Automated Vehicles","date":"2021-08-24","arxiv_id":"2108.10678","n_code_links":0,"syntology":null},{"paper":null,"slug":"relative-entropy-regularized-optimal","title":"Relative Entropy-Regularized Optimal Transport on a Graph: a new algorithm and an experimental comparison","date":"2021-08-23","arxiv_id":"2108.10004","n_code_links":0,"syntology":null},{"paper":"/paper/settling-the-variance-of-multi-agent-policy","slug":"settling-the-variance-of-multi-agent-policy","title":"Settling the Variance of Multi-Agent Policy Gradients","date":"2021-08-19","arxiv_id":"2108.08612","n_code_links":1,"syntology":null},{"paper":"/paper/end-to-end-urban-driving-by-imitating-a","slug":"end-to-end-urban-driving-by-imitating-a","title":"End-to-End Urban Driving by Imitating a Reinforcement Learning Coach","date":"2021-08-18","arxiv_id":"2108.08265","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhejz/carla-roach"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/kitti-carla-a-kitti-like-dataset-generated-by","slug":"kitti-carla-a-kitti-like-dataset-generated-by","title":"KITTI-CARLA: a KITTI-like dataset generated by CARLA Simulator","date":"2021-08-17","arxiv_id":"2109.00892","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-the-robustness-of-semantic","slug":"evaluating-the-robustness-of-semantic","title":"Evaluating the Robustness of Semantic Segmentation for Autonomous Driving against Real-World Adversarial Patch Attacks","date":"2021-08-13","arxiv_id":"2108.06179","n_code_links":1,"syntology":null},{"paper":"/paper/a-functional-mirror-ascent-view-of-policy","slug":"a-functional-mirror-ascent-view-of-policy","title":"A general class of surrogate functions for stable and efficient reinforcement learning","date":"2021-08-12","arxiv_id":"2108.05828","n_code_links":1,"syntology":null},{"paper":null,"slug":"capture-uncertainties-in-deep-neural-networks","title":"Capture Uncertainties in Deep Neural Networks for Safe Operation of Autonomous Driving Vehicles","date":"2021-08-11","arxiv_id":"2108.05118","n_code_links":0,"syntology":null},{"paper":"/paper/carla-a-python-library-to-benchmark","slug":"carla-a-python-library-to-benchmark","title":"CARLA: A Python Library to Benchmark Algorithmic Recourse and Counterfactual Explanation Algorithms","date":"2021-08-02","arxiv_id":"2108.00783","n_code_links":4,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["indyfree/CARLA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"value-based-reinforcement-learning-for","title":"Value-Based Reinforcement Learning for Continuous Control Robotic Manipulation in Multi-Task Sparse Reward Settings","date":"2021-07-28","arxiv_id":"2107.13356","n_code_links":0,"syntology":null},{"paper":"/paper/marsexplorer-exploration-of-unknown-terrains","slug":"marsexplorer-exploration-of-unknown-terrains","title":"MarsExplorer: Exploration of Unknown Terrains via Deep Reinforcement Learning and Procedurally Generated Environments","date":"2021-07-21","arxiv_id":"2107.09996","n_code_links":2,"syntology":null},{"paper":"/paper/improving-exploration-in-policy-gradient","slug":"improving-exploration-in-policy-gradient","title":"Improving exploration in policy gradient search: Application to symbolic optimization","date":"2021-07-19","arxiv_id":"2107.09158","n_code_links":1,"syntology":null},{"paper":"/paper/is-attention-to-bounding-boxes-all-you-need","slug":"is-attention-to-bounding-boxes-all-you-need","title":"Is attention to bounding boxes all you need for pedestrian action prediction?","date":"2021-07-16","arxiv_id":"2107.08031","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-optimal-transport-in-high-dimensions","title":"Scalable Optimal Transport in High Dimensions for Graph Distances, Embedding Alignment, and More","date":"2021-07-14","arxiv_id":"2107.06876","n_code_links":0,"syntology":null},{"paper":"/paper/distributed-online-service-coordination-using","slug":"distributed-online-service-coordination-using","title":"Distributed Online Service Coordination Using Deep Reinforcement Learning","date":"2021-07-07","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"structure-aware-reinforcement-learning-for","title":"Structure-aware reinforcement learning for node-overload protection in mobile edge computing","date":"2021-06-29","arxiv_id":"2107.01025","n_code_links":0,"syntology":null},{"paper":"/paper/brax-a-differentiable-physics-engine-for","slug":"brax-a-differentiable-physics-engine-for","title":"Brax -- A Differentiable Physics Engine for Large Scale Rigid Body Simulation","date":"2021-06-24","arxiv_id":"2106.13281","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google/brax"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/safe-local-motion-planning-with-self","slug":"safe-local-motion-planning-with-self","title":"Safe Local Motion Planning With Self-Supervised Freespace Forecasting","date":"2021-06-19","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"multi-modal-scene-compliant-user-intention","title":"Multi-modal Scene-compliant User Intention Estimation in Navigation","date":"2021-06-13","arxiv_id":"2106.06920","n_code_links":0,"syntology":null},{"paper":null,"slug":"keyframe-focused-visual-imitation-learning","title":"Keyframe-Focused Visual Imitation Learning","date":"2021-06-11","arxiv_id":"2106.06452","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-by-watching","title":"Learning by Watching","date":"2021-06-10","arxiv_id":"2106.05966","n_code_links":0,"syntology":null},{"paper":null,"slug":"don-t-get-yourself-into-trouble-risk-aware","title":"Don't Get Yourself into Trouble! Risk-aware Decision-Making for Autonomous Vehicles","date":"2021-06-08","arxiv_id":"2106.04625","n_code_links":0,"syntology":null},{"paper":null,"slug":"linear-convergence-of-entropy-regularized","title":"Linear Convergence of Entropy-Regularized Natural Policy Gradient with Linear Function Approximation","date":"2021-06-08","arxiv_id":"2106.04096","n_code_links":0,"syntology":null},{"paper":null,"slug":"safe-deep-q-network-for-autonomous-vehicles","title":"Safe Deep Q-Network for Autonomous Vehicles at Unsignalized Intersection","date":"2021-06-08","arxiv_id":"2106.04561","n_code_links":0,"syntology":null},{"paper":null,"slug":"average-reward-reinforcement-learning-with-2","title":"Average-Reward Reinforcement Learning with Trust Region Methods","date":"2021-06-07","arxiv_id":"2106.03442","n_code_links":0,"syntology":null},{"paper":"/paper/cross-trajectory-representation-learning-for","slug":"cross-trajectory-representation-learning-for","title":"Cross-Trajectory Representation Learning for Zero-Shot Generalization in RL","date":"2021-06-04","arxiv_id":"2106.02193","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["bmazoure/ctrl_public"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"distributional-sliced-embedding-discrepancy","title":"Heterogeneous Wasserstein Discrepancy for Incomparable Distributions","date":"2021-06-04","arxiv_id":"2106.02542","n_code_links":0,"syntology":null},{"paper":"/paper/lifetime-policy-reuse-and-the-importance-of","slug":"lifetime-policy-reuse-and-the-importance-of","title":"Lifetime policy reuse and the importance of task capacity","date":"2021-06-03","arxiv_id":"2106.01741","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-entropy-regularization-free-mechanism-for","title":"An Entropy Regularization Free Mechanism for Policy-based Reinforcement Learning","date":"2021-06-01","arxiv_id":"2106.00707","n_code_links":0,"syntology":null},{"paper":null,"slug":"clipping-loops-for-sample-efficient-dialogue","title":"Clipping Loops for Sample-Efficient Dialogue Policy Optimisation","date":"2021-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-policy-extragradient-methods-for","title":"Fast Policy Extragradient Methods for Competitive Games with Entropy Regularization","date":"2021-05-31","arxiv_id":"2105.15186","n_code_links":0,"syntology":null},{"paper":null,"slug":"urban-traffic-surveillance-uts-a-fully","title":"Urban Traffic Surveillance (UTS): A fully probabilistic 3D tracking approach based on 2D detections","date":"2021-05-31","arxiv_id":"2105.14993","n_code_links":0,"syntology":null},{"paper":"/paper/pylot-a-modular-platform-for-exploring-1","slug":"pylot-a-modular-platform-for-exploring-1","title":"Pylot: A Modular Platform for Exploring Latency-Accuracy Tradeoffs in Autonomous Vehicles","date":"2021-05-30","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"shaped-policy-search-for-evolutionary","title":"Shaped Policy Search for Evolutionary Strategies using Waypoints","date":"2021-05-30","arxiv_id":"2105.14639","n_code_links":0,"syntology":null},{"paper":"/paper/sbevnet-end-to-end-deep-stereo-layout-1","slug":"sbevnet-end-to-end-deep-stereo-layout-1","title":"SBEVNet: End-to-End Deep Stereo Layout Estimation","date":"2021-05-25","arxiv_id":"2105.11705","n_code_links":1,"syntology":null},{"paper":"/paper/omni-supervised-point-cloud-segmentation-via","slug":"omni-supervised-point-cloud-segmentation-via","title":"Omni-supervised Point Cloud Segmentation via Gradual Receptive Field Component Reasoning","date":"2021-05-21","arxiv_id":"2105.10203","n_code_links":2,"syntology":{"ran":4,"of":9,"n_ran_checked":4,"n_instrument":0,"unverified":5,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["azuki-miho/RFCR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mctg-multi-frequency-continuous-share-trading","title":"A parallel-network continuous quantitative trading model with GARCH and PPO","date":"2021-05-08","arxiv_id":"2105.03625","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-linear-convergence-of-natural-policy","title":"On the Linear convergence of Natural Policy Gradient Algorithm","date":"2021-05-04","arxiv_id":"2105.01424","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-drive-from-a-world-on-rails","slug":"learning-to-drive-from-a-world-on-rails","title":"Learning to drive from a world on rails","date":"2021-05-03","arxiv_id":"2105.00636","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dotchen/WorldOnRails"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-perception-via-sensor-placement","slug":"improving-perception-via-sensor-placement","title":"Investigating the Impact of Multi-LiDAR Placement on Object Detection for Autonomous Driving","date":"2021-05-02","arxiv_id":"2105.00373","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["HanjiangHu/Multi-LiDAR-Placement-for-3D-Detection"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"network-space-search-for-pareto-efficient","title":"Network Space Search for Pareto-Efficient Spaces","date":"2021-04-22","arxiv_id":"2104.11014","n_code_links":0,"syntology":null},{"paper":"/paper/multi-modal-fusion-transformer-for-end-to-end","slug":"multi-modal-fusion-transformer-for-end-to-end","title":"Multi-Modal Fusion Transformer for End-to-End Autonomous Driving","date":"2021-04-19","arxiv_id":"2104.09224","n_code_links":2,"syntology":{"ran":9,"of":15,"n_ran_checked":8,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"9 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["autonomousvision/transfuser"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"end-to-end-interactive-prediction-and","title":"End-to-End Interactive Prediction and Planning with Optical Flow Distillation for Autonomous Driving","date":"2021-04-18","arxiv_id":"2104.08862","n_code_links":0,"syntology":null},{"paper":"/paper/end-to-end-keyword-spotting-using-neural","slug":"end-to-end-keyword-spotting-using-neural","title":"End-to-end Keyword Spotting using Neural Architecture Search and Quantization","date":"2021-04-14","arxiv_id":"2104.06666","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-mental-models-through-preview-of","title":"Building Mental Models through Preview of Autopilot Behaviors","date":"2021-04-12","arxiv_id":"2104.05470","n_code_links":0,"syntology":null},{"paper":"/paper/a-bayesian-approach-to-reinforcement-learning","slug":"a-bayesian-approach-to-reinforcement-learning","title":"A Bayesian Approach to Reinforcement Learning of Vision-Based Vehicular Control","date":"2021-04-08","arxiv_id":"2104.03807","n_code_links":1,"syntology":null},{"paper":"/paper/a-reinforcement-learning-environment-for-job","slug":"a-reinforcement-learning-environment-for-job","title":"A Reinforcement Learning Environment For Job-Shop Scheduling","date":"2021-04-08","arxiv_id":"2104.03760","n_code_links":4,"syntology":null},{"paper":null,"slug":"risk-aware-lane-selection-on-highway-with","title":"Risk-Aware Lane Selection on Highway with Dynamic Obstacles","date":"2021-04-08","arxiv_id":"2104.04105","n_code_links":0,"syntology":null},{"paper":null,"slug":"progressive-extension-of-reinforcement","title":"Progressive extension of reinforcement learning action dimension for asymmetric assembly tasks","date":"2021-04-06","arxiv_id":"2104.04078","n_code_links":0,"syntology":null},{"paper":"/paper/weakly-supervised-image-semantic-segmentation","slug":"weakly-supervised-image-semantic-segmentation","title":"Weakly-Supervised Image Semantic Segmentation Using Graph Convolutional Networks","date":"2021-03-31","arxiv_id":"2103.16762","n_code_links":1,"syntology":null},{"paper":null,"slug":"flexible-mpc-based-conflict-resolution-using","title":"Flexible MPC-based Conflict Resolution Using Online Adaptive ADMM","date":"2021-03-25","arxiv_id":"2103.14118","n_code_links":0,"syntology":null},{"paper":null,"slug":"hierarchical-program-triggered-reinforcement","title":"Hierarchical Program-Triggered Reinforcement Learning Agents For Automated Driving","date":"2021-03-25","arxiv_id":"2103.13861","n_code_links":0,"syntology":null},{"paper":null,"slug":"convex-online-video-frame-subset-selection","title":"Convex Online Video Frame Subset Selection using Multiple Criteria for Data Efficient Autonomous Driving","date":"2021-03-24","arxiv_id":"2103.13021","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-steering-angle-prediction-for","title":"Self-Supervised Steering Angle Prediction for Vehicle Control Using Visual Odometry","date":"2021-03-20","arxiv_id":"2103.11204","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-energy-saving-snake-locomotion-gait-policy","title":"An Energy-Saving Snake Locomotion Gait Policy Obtained Using Deep Reinforcement Learning","date":"2021-03-08","arxiv_id":"2103.04511","n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-explanation-using-attention-mechanism-1","title":"Visual Explanation using Attention Mechanism in Actor-Critic-based Deep Reinforcement Learning","date":"2021-03-06","arxiv_id":"2103.04067","n_code_links":0,"syntology":null},{"paper":"/paper/the-surprising-effectiveness-of-mappo-in","slug":"the-surprising-effectiveness-of-mappo-in","title":"The Surprising Effectiveness of PPO in Cooperative, Multi-Agent Games","date":"2021-03-02","arxiv_id":"2103.01955","n_code_links":19,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marlbenchmark/on-policy"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/a-deeppixbis-attentional-angular-margin-for","slug":"a-deeppixbis-attentional-angular-margin-for","title":"A-DeepPixBis: Attentional Angular Margin for Face Anti-Spoofing","date":"2021-03-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"autopreview-a-framework-for-autopilot","title":"AutoPreview: A Framework for Autopilot Behavior Understanding","date":"2021-02-25","arxiv_id":"2102.13034","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatio-temporal-look-ahead-trajectory","title":"Spatio-Temporal Look-Ahead Trajectory Prediction using Memory Neural Network","date":"2021-02-24","arxiv_id":"2102.12070","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-proximal-policy-optimization-s-heavy-1","title":"On Proximal Policy Optimization's Heavy-tailed Gradients","date":"2021-02-20","arxiv_id":"2102.10264","n_code_links":0,"syntology":null},{"paper":"/paper/combining-events-and-frames-using-recurrent","slug":"combining-events-and-frames-using-recurrent","title":"Combining Events and Frames using Recurrent Asynchronous Multimodal Networks for Monocular Depth Prediction","date":"2021-02-18","arxiv_id":"2102.09320","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["uzh-rpg/rpg_ramnet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/domain-adaptation-in-reinforcement-learning","slug":"domain-adaptation-in-reinforcement-learning","title":"Domain Adaptation In Reinforcement Learning Via Latent Unified State Representation","date":"2021-02-10","arxiv_id":"2102.05714","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["KarlXing/LUSR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-survey-of-motion-planning-algorithms-for","title":"A review of motion planning algorithms for intelligent robotics","date":"2021-02-04","arxiv_id":"2102.02376","n_code_links":0,"syntology":null},{"paper":null,"slug":"affordance-based-reinforcement-learning-for","title":"Affordance-based Reinforcement Learning for Urban Driving","date":"2021-01-15","arxiv_id":"2101.05970","n_code_links":0,"syntology":null},{"paper":"/paper/instance-aware-predictive-navigation-in-multi","slug":"instance-aware-predictive-navigation-in-multi","title":"Instance-Aware Predictive Navigation in Multi-Agent Environments","date":"2021-01-14","arxiv_id":"2101.05893","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-strong-on-policy-competitor-to-ppo","title":"A Strong On-Policy Competitor To PPO","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-multi-agent-deep-reinforcement","title":"Benchmarking Multi-Agent Deep Reinforcement Learning Algorithms","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-coherent-exploration-for-continuous","title":"Deep Coherent Exploration For Continuous Control","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-mnas-uncertainty-aware-neural","title":"Fast MNAS: Uncertainty-aware Neural Architecture Search with Lifelong Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"grounded-compositional-generalization-with","title":"Grounded Compositional Generalization with Environment Interactions","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"inverse-reinforcement-learning-for-autonomous","title":"Inverse reinforcement learning for autonomous navigation via differentiable semantic mapping and planning","date":"2021-01-01","arxiv_id":"2101.00186","n_code_links":0,"syntology":null},{"paper":null,"slug":"optimizing-information-bottleneck-in","title":"Optimizing Information Bottleneck in Reinforcement Learning: A Stein Variational Approach","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pgps-coupling-policy-gradient-with-population","title":"PGPS : Coupling Policy Gradient with Population-based Search","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-optimization-in-zero-sum-markov-games","title":"Policy Optimization in Zero-Sum Markov Games: Fictitious Self-Play Provably Attains Nash Equilibria","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"tmcoss-thresholded-multi-criteria-online","title":"TMCOSS: Thresholded Multi-Criteria Online Subset Selection for Data-Efficient Autonomous Driving","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"warpspeed-computation-of-optimal-transport","title":"Warpspeed Computation of Optimal Transport, Graph Distances, and Embedding Alignment","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"asynchronous-advantage-actor-critic-non-1","title":"Towards Understanding Asynchronous Advantage Actor-critic: Convergence and Linear Speedup","date":"2020-12-31","arxiv_id":"2012.15511","n_code_links":0,"syntology":null},{"paper":null,"slug":"reconfigurable-intelligent-surface-assisted-6","title":"Reconfigurable Intelligent Surface Assisted Mobile Edge Computing with Heterogeneous Learning Tasks","date":"2020-12-25","arxiv_id":"2012.13533","n_code_links":0,"syntology":null},{"paper":null,"slug":"2012-11643","title":"myGym: Modular Toolkit for Visuomotor Robotic Tasks","date":"2020-12-21","arxiv_id":"2012.11643","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modal-depth-estimation-using","title":"Multi-Modal Depth Estimation Using Convolutional Neural Networks","date":"2020-12-17","arxiv_id":"2012.09667","n_code_links":0,"syntology":null},{"paper":null,"slug":"carla-real-traffic-scenarios-novel-training","title":"CARLA Real Traffic Scenarios -- novel training ground and benchmark for autonomous driving","date":"2020-12-16","arxiv_id":"2012.11329","n_code_links":0,"syntology":null},{"paper":"/paper/increasing-data-efficiency-of-driving-agent","slug":"increasing-data-efficiency-of-driving-agent","title":"Increasing Data Efficiency of Driving Agent By World Model","date":"2020-12-14","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"ipm-move-planner-an-efficient-exploiting-deep","title":"IPM Move Planner: AN EFFICIENT EXPLOITING DEEP REINFORCEMENT LEARNING WITH MONTE CARLO TREE SEARCH","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mobile-robots-exploration-via-deep","title":"Mobile Robots Exploration via Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"policy-gradient-for-items-recommendation-on","title":"Policy Gradient for items Recommendation on Virtual Taobao","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","n_code_links":1,"syntology":null},{"paper":null,"slug":"train-a-snake-with-reinforcement-learning","title":"Train a snake with reinforcement learning algorithms","date":"2020-12-14","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/simple-copy-paste-is-a-strong-data","slug":"simple-copy-paste-is-a-strong-data","title":"Simple Copy-Paste is a Strong Data Augmentation Method for Instance Segmentation","date":"2020-12-13","arxiv_id":"2012.07177","n_code_links":5,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tensorflow/tpu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/re-reimplementation-of-fixmatch-and","slug":"re-reimplementation-of-fixmatch-and","title":"[Re] Reimplementation of FixMatch and Investigation on Noisy (Pseudo) Labels and Confirmation Errors of FixMatch","date":"2020-12-06","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-smoothed","title":"Proximal Policy Optimization Smoothed Algorithm","date":"2020-12-04","arxiv_id":"2012.02439","n_code_links":0,"syntology":null},{"paper":"/paper/domain-generalization-via-entropy","slug":"domain-generalization-via-entropy","title":"Domain Generalization via Entropy Regularization","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-convergence-of-smooth-regularized","title":"On the Convergence of Smooth Regularized Approximate Value Iteration Schemes","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"promoting-stochasticity-for-expressive","title":"Promoting Stochasticity for Expressive Policies via a Simple and Efficient Regularization Method","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/an-end-to-end-deep-reinforcement-learning","slug":"an-end-to-end-deep-reinforcement-learning","title":"An End-to-end Deep Reinforcement Learning Approach for the Long-term Short-term Planning on the Frenet Space","date":"2020-11-26","arxiv_id":"2011.13098","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhanced-scene-specificity-with-sparse","title":"Enhanced Scene Specificity with Sparse Dynamic Value Estimation","date":"2020-11-25","arxiv_id":"2011.12574","n_code_links":0,"syntology":null},{"paper":"/paper/finrl-a-deep-reinforcement-learning-library","slug":"finrl-a-deep-reinforcement-learning-library","title":"FinRL: A Deep Reinforcement Learning Library for Automated Stock Trading in Quantitative Finance","date":"2020-11-19","arxiv_id":"2011.09607","n_code_links":6,"syntology":null},{"paper":"/paper/attentivenas-improving-neural-architecture","slug":"attentivenas-improving-neural-architecture","title":"AttentiveNAS: Improving Neural Architecture Search via Attentive Sampling","date":"2020-11-18","arxiv_id":"2011.09011","n_code_links":2,"syntology":null},{"paper":"/paper/is-independent-learning-all-you-need-in-the","slug":"is-independent-learning-all-you-need-in-the","title":"Is Independent Learning All You Need in the StarCraft Multi-Agent Challenge?","date":"2020-11-18","arxiv_id":"2011.09533","n_code_links":7,"syntology":null}],"record_sha256":"537f3ecb03228e25d14f96b7c3ccecdbca3d212717a708e4c4fbafed6ddc2cee","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}