{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/entropy-regularization/papers/8","list_of":"/method/entropy-regularization","method":"Entropy Regularization","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":8,"pages_in_order":12,"rows_per_page":100,"rows":[701,800],"of":1128,"counts":{"archive_papers_tagged":1128,"with_a_code_link":451,"where_syntology_ran_a_sample":156,"not_listed_spam_title":0,"listed":1128,"listed_where_code_ran":156,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":129,"every_run_a_failure_of_syntologys_instrument":27,"listed_with_a_run_with_no_instrument_failure":129,"listed_every_run_a_failure_of_syntologys_instrument":27,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/entropy-regularization","prev":"/method/entropy-regularization/papers/7","next":"/method/entropy-regularization/papers/9","papers":[{"paper":null,"slug":"accelerating-federated-edge-learning-via-1","title":"Accelerating Federated Edge Learning via Topology Optimization","date":"2022-04-01","arxiv_id":"2204.00489","n_code_links":0,"syntology":null},{"paper":"/paper/hysteresis-based-rl-robustifying","slug":"hysteresis-based-rl-robustifying","title":"Hysteresis-Based RL: Robustifying Reinforcement Learning-based Control Policies via Hybrid Control","date":"2022-04-01","arxiv_id":"2204.00654","n_code_links":2,"syntology":null},{"paper":null,"slug":"semi-fairvae-semi-supervised-fair","title":"Semi-FairVAE: Semi-supervised Fair Representation Learning with Adversarial Variational Autoencoder","date":"2022-04-01","arxiv_id":"2204.00536","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-word-error-rate-a-good-evaluation-metric","title":"Is Word Error Rate a good evaluation metric for Speech Recognition in Indic Languages?","date":"2022-03-30","arxiv_id":"2203.16601","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-evolutionary-terrain-generation","title":"Assessing Evolutionary Terrain Generation Methods for Curriculum Reinforcement Learning","date":"2022-03-29","arxiv_id":"2203.15172","n_code_links":0,"syntology":null},{"paper":null,"slug":"your-policy-regularizer-is-secretly-an","title":"Your Policy Regularizer is Secretly an Adversary","date":"2022-03-23","arxiv_id":"2203.12592","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-all-vehicles","slug":"learning-from-all-vehicles","title":"Learning from All Vehicles","date":"2022-03-22","arxiv_id":"2203.11934","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dotchen/LAV"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"linear-convergence-of-a-policy-gradient","title":"Linear convergence of a policy gradient method for some finite horizon continuous time control problems","date":"2022-03-22","arxiv_id":"2203.11758","n_code_links":0,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-based-transmit","title":"Proximal Policy Optimization-based Transmit Beamforming and Phase-shift Design in an IRS-aided ISAC System for the THz Band","date":"2022-03-21","arxiv_id":"2203.10819","n_code_links":0,"syntology":null},{"paper":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","n_code_links":1,"syntology":null},{"paper":"/paper/v2x-vit-vehicle-to-everything-cooperative","slug":"v2x-vit-vehicle-to-everything-cooperative","title":"V2X-ViT: Vehicle-to-Everything Cooperative Perception with Vision Transformer","date":"2022-03-20","arxiv_id":"2203.10638","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DerrickXuNu/v2x-vit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"dta-physical-camouflage-attacks-using","title":"DTA: Physical Camouflage Attacks using Differentiable Transformation Network","date":"2022-03-18","arxiv_id":"2203.09831","n_code_links":0,"syntology":null},{"paper":null,"slug":"proximal-policy-optimization-with-adaptive","title":"Proximal Policy Optimization with Adaptive Threshold for Symmetric Relative Density Ratio","date":"2022-03-18","arxiv_id":"2203.09809","n_code_links":0,"syntology":null},{"paper":null,"slug":"conquering-ghosts-relation-learning-for","title":"Conquering Ghosts: Relation Learning for Information Reliability Representation and End-to-End Robust Navigation","date":"2022-03-14","arxiv_id":"2203.09952","n_code_links":0,"syntology":null},{"paper":"/paper/synwoodscape-synthetic-surround-view-fisheye","slug":"synwoodscape-synthetic-surround-view-fisheye","title":"SynWoodScape: Synthetic Surround-view Fisheye Camera Dataset for Autonomous Driving","date":"2022-03-09","arxiv_id":"2203.05056","n_code_links":0,"syntology":null},{"paper":"/paper/mirror-differentiable-deep-social-projection","slug":"mirror-differentiable-deep-social-projection","title":"MIRROR: Differentiable Deep Social Projection for Assistive Human-Robot Communication","date":"2022-03-06","arxiv_id":"2203.02877","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-aided-traffic-control-scheme-for-m2m","title":"AI-aided Traffic Control Scheme for M2M Communications in the Internet of Vehicles","date":"2022-03-05","arxiv_id":"2204.03504","n_code_links":0,"syntology":null},{"paper":"/paper/risk-aware-scene-sampling-for-dynamic","slug":"risk-aware-scene-sampling-for-dynamic","title":"Risk-Aware Scene Sampling for Dynamic Assurance of Autonomous Systems","date":"2022-02-28","arxiv_id":"2202.13510","n_code_links":1,"syntology":null},{"paper":"/paper/panoflow-learning-optical-flow-for-panoramic","slug":"panoflow-learning-optical-flow-for-panoramic","title":"PanoFlow: Learning 360° Optical Flow for Surrounding Temporal Understanding","date":"2022-02-27","arxiv_id":"2202.13388","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["masterhow/panoflow"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"context-hierarchy-inverse-reinforcement","title":"Context-Hierarchy Inverse Reinforcement Learning","date":"2022-02-25","arxiv_id":"2202.12597","n_code_links":0,"syntology":null},{"paper":null,"slug":"consistent-dropout-for-policy-gradient","title":"Consistent Dropout for Policy Gradient Reinforcement Learning","date":"2022-02-23","arxiv_id":"2202.11818","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-eye-driving-with-the-eyes-of-ai-for-corner","title":"A-Eye: Driving with the Eyes of AI for Corner Case Generation","date":"2022-02-22","arxiv_id":"2202.10803","n_code_links":0,"syntology":null},{"paper":"/paper/a-self-supervised-descriptor-for-image-copy","slug":"a-self-supervised-descriptor-for-image-copy","title":"A Self-Supervised Descriptor for Image Copy Detection","date":"2022-02-21","arxiv_id":"2202.10261","n_code_links":2,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["facebookresearch/sscd-copy-detection"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/don-t-touch-what-matters-task-aware-lipschitz","slug":"don-t-touch-what-matters-task-aware-lipschitz","title":"Don't Touch What Matters: Task-Aware Lipschitz Data Augmentation for Visual Reinforcement Learning","date":"2022-02-21","arxiv_id":"2202.09982","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-a-shield-from-catastrophic-action","title":"Learning a Shield from Catastrophic Action Effects: Never Repeat the Same Mistake","date":"2022-02-19","arxiv_id":"2202.09516","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-task-safe-reinforcement-learning-for","title":"Multi-task Safe Reinforcement Learning for Navigating Intersections in Dense Traffic","date":"2022-02-19","arxiv_id":"2202.09644","n_code_links":0,"syntology":null},{"paper":"/paper/cadre-a-cascade-deep-reinforcement-learning","slug":"cadre-a-cascade-deep-reinforcement-learning","title":"CADRE: A Cascade Deep Reinforcement Learning Framework for Vision-based Autonomous Urban Driving","date":"2022-02-17","arxiv_id":"2202.08557","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["BIT-MCS/Cadre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"omnisyn-synthesizing-360-videos-with-wide","title":"OmniSyn: Synthesizing 360 Videos with Wide-baseline Panoramas","date":"2022-02-17","arxiv_id":"2202.08752","n_code_links":0,"syntology":null},{"paper":"/paper/multi-modal-fusion-for-sensorimotor","slug":"multi-modal-fusion-for-sensorimotor","title":"Multi-Modal Fusion for Sensorimotor Coordination in Steering Angle Prediction","date":"2022-02-11","arxiv_id":"2202.05500","n_code_links":1,"syntology":null},{"paper":null,"slug":"empirical-risk-minimization-with-relative","title":"Empirical Risk Minimization with Relative Entropy Regularization: Optimality and Sensitivity Analysis","date":"2022-02-09","arxiv_id":"2202.04385","n_code_links":0,"syntology":null},{"paper":null,"slug":"revisiting-qmix-discriminative-credit","title":"Revisiting QMIX: Discriminative Credit Assignment by Gradient Entropy Regularization","date":"2022-02-09","arxiv_id":"2202.04427","n_code_links":0,"syntology":null},{"paper":"/paper/skrl-modular-and-flexible-library-for","slug":"skrl-modular-and-flexible-library-for","title":"skrl: Modular and Flexible Library for Reinforcement Learning","date":"2022-02-08","arxiv_id":"2202.03825","n_code_links":1,"syntology":null},{"paper":null,"slug":"simulation-to-reality-domain-adaptation-for","title":"Simulation-to-Reality domain adaptation for offline 3D object annotation on pointclouds with correlation alignment","date":"2022-02-06","arxiv_id":"2202.02666","n_code_links":0,"syntology":null},{"paper":null,"slug":"ai-as-a-service-toolkit-for-human-centered","title":"AI-as-a-Service Toolkit for Human-Centered Intelligence in Autonomous Driving","date":"2022-02-03","arxiv_id":"2202.01645","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-machine-learning-smartphone-based-sensing","title":"A Machine Learning Smartphone-based Sensing for Driver Behavior Classification","date":"2022-02-01","arxiv_id":"2202.01893","n_code_links":0,"syntology":null},{"paper":null,"slug":"monotonic-improvement-guarantees-under-non-1","title":"Trust Region Bounds for Decentralized PPO Under Non-stationarity","date":"2022-01-31","arxiv_id":"2202.00082","n_code_links":0,"syntology":null},{"paper":null,"slug":"you-may-not-need-ratio-clipping-in-ppo","title":"You May Not Need Ratio Clipping in PPO","date":"2022-01-31","arxiv_id":"2202.00079","n_code_links":0,"syntology":null},{"paper":"/paper/do-you-need-the-entropy-reward-in-practice","slug":"do-you-need-the-entropy-reward-in-practice","title":"Do You Need the Entropy Reward (in Practice)?","date":"2022-01-28","arxiv_id":"2201.12434","n_code_links":2,"syntology":null},{"paper":null,"slug":"neuro-symbolic-entropy-regularization","title":"Neuro-Symbolic Entropy Regularization","date":"2022-01-25","arxiv_id":"2201.11250","n_code_links":0,"syntology":null},{"paper":null,"slug":"ray-based-distributed-autonomous-vehicle","title":"Ray Based Distributed Autonomous Vehicle Research Platform","date":"2022-01-18","arxiv_id":"2201.06835","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatiotemporal-costmap-inference-for-mpc-via","title":"Spatiotemporal Costmap Inference for MPC via Deep Inverse Reinforcement Learning","date":"2022-01-17","arxiv_id":"2201.06539","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-37-implementation-details-of-proximal","title":"The 37 Implementation Details of Proximal Policy Optimization","date":"2022-01-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-study-on-mitigating-hard-boundaries-of","title":"A Study on Mitigating Hard Boundaries of Decision-Tree-based Uncertainty Estimates for AI Models","date":"2022-01-10","arxiv_id":"2201.03263","n_code_links":0,"syntology":null},{"paper":"/paper/mirror-learning-a-unifying-framework-of","slug":"mirror-learning-a-unifying-framework-of","title":"Mirror Learning: A Unifying Framework of Policy Optimisation","date":"2022-01-07","arxiv_id":"2201.02373","n_code_links":1,"syntology":null},{"paper":"/paper/dreyevr-democratizing-driving-simulation-in","slug":"dreyevr-democratizing-driving-simulation-in","title":"DReyeVR: Democratizing Virtual Reality Driving Simulation for Behavioural & Interaction Research","date":"2022-01-06","arxiv_id":"2201.01931","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-robustness-of-neural-networks","title":"Towards Robustness of Neural Networks","date":"2021-12-30","arxiv_id":"2112.15188","n_code_links":0,"syntology":null},{"paper":null,"slug":"ddpg-car-following-model-with-real-world","title":"Modified DDPG car-following model with a real-world human driving experience with CARLA simulator","date":"2021-12-29","arxiv_id":"2112.14602","n_code_links":0,"syntology":null},{"paper":null,"slug":"2112-13937","title":"Multiagent Model-based Credit Assignment for Continuous Control","date":"2021-12-27","arxiv_id":"2112.13937","n_code_links":0,"syntology":null},{"paper":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","n_code_links":1,"syntology":null},{"paper":null,"slug":"doppler-velocity-based-algorithm-for","title":"Doppler velocity-based algorithm for Clustering and Velocity Estimation of moving objects","date":"2021-12-24","arxiv_id":"2112.12984","n_code_links":0,"syntology":null},{"paper":"/paper/intersection-focused-situation-coverage-based","slug":"intersection-focused-situation-coverage-based","title":"Intersection focused Situation Coverage-based Verification and Validation Framework for Autonomous Vehicles Implemented in CARLA","date":"2021-12-24","arxiv_id":"2112.14706","n_code_links":1,"syntology":null},{"paper":"/paper/maximum-entropy-population-based-training-for-1","slug":"maximum-entropy-population-based-training-for-1","title":"Maximum Entropy Population-Based Training for Zero-Shot Human-AI Coordination","date":"2021-12-22","arxiv_id":"2112.11701","n_code_links":3,"syntology":null},{"paper":null,"slug":"a-deep-reinforcement-learning-model-for","title":"A deep reinforcement learning model for predictive maintenance planning of road assets: Integrating LCA and LCCA","date":"2021-12-20","arxiv_id":"2112.12589","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-reward-machines-a-study-in-partially","title":"Learning Reward Machines: A Study in Partially Observable Reinforcement Learning","date":"2021-12-17","arxiv_id":"2112.09477","n_code_links":0,"syntology":null},{"paper":null,"slug":"scientific-discovery-and-the-cost-of","title":"Scientific Discovery and the Cost of Measurement -- Balancing Information and Cost in Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07535","n_code_links":0,"syntology":null},{"paper":"/paper/real-time-collision-risk-estimation-based-on","slug":"real-time-collision-risk-estimation-based-on","title":"Real-time Collision Risk Estimation based on Stochastic Reachability Spaces","date":"2021-12-08","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"ptr-ppo-proximal-policy-optimization-with","title":"PTR-PPO: Proximal Policy Optimization with Prioritized Trajectory Replay","date":"2021-12-07","arxiv_id":"2112.03798","n_code_links":0,"syntology":null},{"paper":null,"slug":"personalized-federated-learning-of-driver","title":"Personalized Federated Learning of Driver Prediction Models for Autonomous Driving","date":"2021-12-02","arxiv_id":"2112.00956","n_code_links":0,"syntology":null},{"paper":null,"slug":"paris-carla-3d-a-real-and-synthetic-outdoor","title":"Paris-CARLA-3D: A Real and Synthetic Outdoor Point Cloud Dataset for Challenging Tasks in 3D Mapping","date":"2021-11-22","arxiv_id":"2111.11348","n_code_links":0,"syntology":null},{"paper":"/paper/learning-robust-output-control-barrier","slug":"learning-robust-output-control-barrier","title":"Learning Robust Output Control Barrier Functions from Safe Expert Demonstrations","date":"2021-11-18","arxiv_id":"2111.09971","n_code_links":1,"syntology":null},{"paper":"/paper/gri-general-reinforced-imitation-and-its","slug":"gri-general-reinforced-imitation-and-its","title":"GRI: General Reinforced Imitation and its Application to Vision-Based Autonomous Driving","date":"2021-11-16","arxiv_id":"2111.08575","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-pseudo-projection-operator-applications","title":"The Pseudo Projection Operator: Applications of Deep Learning to Projection Based Filtering in Non-Trivial Frequency Regimes","date":"2021-11-13","arxiv_id":"2111.07140","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparison-of-model-free-and-model","title":"A Comparison of Model-Free and Model Predictive Control for Price Responsive Water Heaters","date":"2021-11-08","arxiv_id":"2111.04689","n_code_links":0,"syntology":null},{"paper":"/paper/coordinated-proximal-policy-optimization","slug":"coordinated-proximal-policy-optimization","title":"Coordinated Proximal Policy Optimization","date":"2021-11-07","arxiv_id":"2111.04051","n_code_links":1,"syntology":null},{"paper":null,"slug":"ai-based-radio-resource-management-and","title":"AI-based Radio Resource Management and Trajectory Design for PD-NOMA Communication in IRS-UAV Assisted Networks","date":"2021-11-06","arxiv_id":"2111.03869","n_code_links":0,"syntology":null},{"paper":null,"slug":"tnd-nas-towards-non-differentiable-objectives","title":"TND-NAS: Towards Non-differentiable Objectives in Progressive Differentiable NAS Framework","date":"2021-11-06","arxiv_id":"2111.03892","n_code_links":0,"syntology":null},{"paper":"/paper/compressing-sensor-data-for-remote-assistance","slug":"compressing-sensor-data-for-remote-assistance","title":"Compressing Sensor Data for Remote Assistance of Autonomous Vehicles using Deep Generative Models","date":"2021-11-05","arxiv_id":"2111.03201","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-rna-secondary-structure-design","title":"Improving RNA Secondary Structure Design using Deep Reinforcement Learning","date":"2021-11-05","arxiv_id":"2111.04504","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-entropic-regularization-in-gans","title":"Understanding Entropic Regularization in GANs","date":"2021-11-02","arxiv_id":"2111.01387","n_code_links":0,"syntology":null},{"paper":"/paper/learning-distilled-collaboration-graph-for","slug":"learning-distilled-collaboration-graph-for","title":"Learning Distilled Collaboration Graph for Multi-Agent Perception","date":"2021-11-01","arxiv_id":"2111.00643","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","official":{"repos":["ai4ce/DiscoNet"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/object-aware-regularization-for-addressing","slug":"object-aware-regularization-for-addressing","title":"Object-Aware Regularization for Addressing Causal Confusion in Imitation Learning","date":"2021-10-27","arxiv_id":"2110.14118","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alinlab/oreo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"entrpo-trust-region-policy-optimization","title":"EnTRPO: Trust Region Policy Optimization Method with Entropy Regularization","date":"2021-10-26","arxiv_id":"2110.13373","n_code_links":0,"syntology":null},{"paper":"/paper/learning-collaborative-policies-to-solve-np","slug":"learning-collaborative-policies-to-solve-np","title":"Learning Collaborative Policies to Solve NP-hard Routing Problems","date":"2021-10-26","arxiv_id":"2110.13987","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":4,"n_instrument":2,"unverified":5,"pointer_only":11,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","official":{"repos":["alstn12088/lcp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-deep-reinforcement-learning-approach-for-9","title":"A Deep Reinforcement Learning Approach for Audio-based Navigation and Audio Source Localization in Multi-speaker Environments","date":"2021-10-25","arxiv_id":"2110.12778","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-distributed-deep-reinforcement-learning","title":"A Distributed Deep Reinforcement Learning Technique for Application Placement in Edge and Fog Computing Environments","date":"2021-10-24","arxiv_id":"2110.12415","n_code_links":0,"syntology":null},{"paper":null,"slug":"cim-ppo-proximal-policy-optimization-with-liu","title":"CIM-PPO:Proximal Policy Optimization with Liu-Correntropy Induced Metric","date":"2021-10-20","arxiv_id":"2110.10522","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-adversarial-imitation-learning-for-1","title":"Generative Adversarial Imitation Learning for End-to-End Autonomous Driving on Urban Environments","date":"2021-10-16","arxiv_id":"2110.08586","n_code_links":0,"syntology":null},{"paper":"/paper/improving-the-sample-efficiency-of-neural","slug":"improving-the-sample-efficiency-of-neural","title":"Improving the sample-efficiency of neural architecture search with reinforcement learning","date":"2021-10-13","arxiv_id":"2110.06751","n_code_links":1,"syntology":null},{"paper":null,"slug":"navigation-in-urban-environments-amongst","title":"Navigation In Urban Environments Amongst Pedestrians Using Multi-Objective Deep Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.05205","n_code_links":0,"syntology":null},{"paper":"/paper/camera-calibration-through-camera-projection","slug":"camera-calibration-through-camera-projection","title":"Camera Calibration through Camera Projection Loss","date":"2021-10-07","arxiv_id":"2110.03479","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-understanding-distributional","title":"The Benefits of Being Categorical Distributional: Uncertainty-aware Regularized Exploration in Reinforcement Learning","date":"2021-10-07","arxiv_id":"2110.03155","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-new-weakly-supervised-approach-for-als","title":"A new weakly supervised approach for ALS point cloud semantic segmentation","date":"2021-10-04","arxiv_id":"2110.01462","n_code_links":0,"syntology":null},{"paper":null,"slug":"cycle-consistent-world-models-for-domain","title":"Cycle-Consistent World Models for Domain Independent Latent Imagination","date":"2021-10-02","arxiv_id":"2110.00808","n_code_links":0,"syntology":null},{"paper":null,"slug":"bitcoin-transaction-strategy-construction","title":"Bitcoin Transaction Strategy Construction Based on Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.14789","n_code_links":0,"syntology":null},{"paper":null,"slug":"fight-fire-with-fire-countering-bad-shortcuts","title":"Fight fire with fire: countering bad shortcuts in imitation learning with good shortcuts","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-maximum-entropy-reinforcement","title":"Generalized Maximum Entropy Reinforcement Learning via Reward Shaping","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"joint-self-supervised-learning-for-vision","title":"Joint Self-Supervised Learning for Vision-based Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"p4o-efficient-deep-reinforcement-learning","title":"P4O: Efficient Deep Reinforcement Learning with Predictive Processing Proximal Policy Optimization","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sample-efficient-stochastic-policy","title":"Sample Efficient Stochastic Policy Extragradient Algorithm for Zero-Sum Markov Game","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-step-towards-efficient-evaluation-of","title":"A Step Towards Efficient Evaluation of Complex Perception Tasks in Simulation","date":"2021-09-28","arxiv_id":"2110.02739","n_code_links":0,"syntology":null},{"paper":"/paper/fast-nonlinear-risk-assessment-for-autonomous","slug":"fast-nonlinear-risk-assessment-for-autonomous","title":"Fast nonlinear risk assessment for autonomous vehicles using learned conditional probabilistic models of agent futures","date":"2021-09-21","arxiv_id":"2109.09975","n_code_links":1,"syntology":null},{"paper":null,"slug":"stochastic-mpc-with-multi-modal-predictions","title":"Stochastic MPC with Multi-modal Predictions for Traffic Intersections","date":"2021-09-20","arxiv_id":"2109.09792","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-state-representation-learning-for","slug":"efficient-state-representation-learning-for","title":"POAR: Efficient Policy Optimization via Online Abstract State Representation Learning","date":"2021-09-17","arxiv_id":"2109.08642","n_code_links":1,"syntology":null},{"paper":"/paper/opv2v-an-open-benchmark-dataset-and-fusion","slug":"opv2v-an-open-benchmark-dataset-and-fusion","title":"OPV2V: An Open Benchmark Dataset and Fusion Pipeline for Perception with Vehicle-to-Vehicle Communication","date":"2021-09-16","arxiv_id":"2109.07644","n_code_links":2,"syntology":null},{"paper":null,"slug":"border-seggcn-improving-semantic-segmentation","title":"Border-SegGCN: Improving Semantic Segmentation by Refining the Border Outline using Graph Convolutional Network","date":"2021-09-11","arxiv_id":"2109.05353","n_code_links":0,"syntology":null},{"paper":"/paper/neat-neural-attention-fields-for-end-to-end","slug":"neat-neural-attention-fields-for-end-to-end","title":"NEAT: Neural Attention Fields for End-to-End Autonomous Driving","date":"2021-09-09","arxiv_id":"2109.04456","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["autonomousvision/neat"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/macrpo-multi-agent-cooperative-recurrent","slug":"macrpo-multi-agent-cooperative-recurrent","title":"MACRPO: Multi-Agent Cooperative Recurrent Policy Optimization","date":"2021-09-02","arxiv_id":"2109.00882","n_code_links":1,"syntology":null},{"paper":"/paper/roadscene2vec-a-tool-for-extracting-and","slug":"roadscene2vec-a-tool-for-extracting-and","title":"roadscene2vec: A Tool for Extracting and Embedding Road Scene-Graphs","date":"2021-09-02","arxiv_id":"2109.01183","n_code_links":1,"syntology":null},{"paper":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"wad-a-deep-reinforcement-learning-agent-for","title":"WAD: A Deep Reinforcement Learning Agent for Urban Autonomous Driving","date":"2021-08-27","arxiv_id":"2108.12134","n_code_links":0,"syntology":null}],"record_sha256":"83eef3e29bae1809a3d1dc4afea4b6fd0017bd93dd5a33097c8999e39a4af0a7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}