{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/imitation-learning/papers/18","list_of":"/task/imitation-learning","task":"Imitation Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":22,"rows_per_page":100,"rows":[1701,1800],"of":2122,"counts":{"archive_papers_tagged":2122,"with_a_code_link":691,"where_syntology_ran_a_sample":234,"not_listed_spam_title":0,"listed":2122,"listed_where_code_ran":234,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":191,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":191,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/imitation-learning","prev":"/task/imitation-learning/papers/17","next":"/task/imitation-learning/papers/19","papers":[{"url":null,"slug":"robust-imitation-via-decision-time-planning","title":"Robust Imitation via Decision-Time Planning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-bayesian-inverse-reinforcement","title":"Scalable Bayesian Inverse Reinforcement Learning by Auto-Encoding Reward","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-motivated-communication-agent-for-real","title":"Self-Motivated Communication Agent for Real-World Vision-Dialog Navigation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unbiased-learning-with-state-conditioned","title":"Unbiased learning with State-Conditioned Rewards in Adversarial Imitation Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-control-with-graph-neural","title":"Synthesizing Decentralized Controllers with Graph Neural Networks and Imitation Learning","date":"2020-12-29","arxiv_id":"2012.14906","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-for-high-precision-peg-in","title":"Imitation Learning for High Precision Peg-in-Hole Tasks","date":"2020-12-26","arxiv_id":"2101.01052","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-instance-aware-localization-for-end-to","title":"Multi-Instance Aware Localization for End-to-End Imitation Learning","date":"2020-12-26","arxiv_id":"2101.01053","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-action-prediction-for-imitation","title":"Stochastic Action Prediction for Imitation Learning","date":"2020-12-26","arxiv_id":"2101.01055","repositories_listed":0,"syntology":null},{"url":null,"slug":"translating-natural-language-instructions-to","title":"Translating Natural Language Instructions to Computer Programs for Robot Manipulation","date":"2020-12-26","arxiv_id":"2012.13695","repositories_listed":0,"syntology":null},{"url":null,"slug":"scc-an-efficient-deep-reinforcement-learning","title":"SCC: an efficient deep reinforcement learning agent mastering the game of StarCraft II","date":"2020-12-24","arxiv_id":"2012.13169","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethink-ai-based-power-grid-control-diving","title":"Rethink AI-based Power Grid Control: Diving Into Algorithm Design","date":"2020-12-23","arxiv_id":"2012.13026","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-imitation-advantage-learning","title":"Self-Imitation Advantage Learning","date":"2020-12-22","arxiv_id":"2012.11989","repositories_listed":0,"syntology":null},{"url":null,"slug":"2012-11643","title":"myGym: Modular Toolkit for Visuomotor Robotic Tasks","date":"2020-12-21","arxiv_id":"2012.11643","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-hierarchical-imitation-and","title":"Active Hierarchical Imitation and Reinforcement Learning","date":"2020-12-14","arxiv_id":"2012.07330","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-play-tetris-with-deep-reinforcement","title":"Learn to Play Tetris with Deep Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-enhanced-gaussian-cross-entropy-in","title":"Using Enhanced Gaussian Cross-Entropy in Imitation Learning to Digging the First Diamond in Minecraft","date":"2020-12-14","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-imitation-learning-using","title":"Human-in-the-Loop Imitation Learning using Remote Teleoperation","date":"2020-12-12","arxiv_id":"2012.06733","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-multi-arm-manipulation-through","title":"Learning Multi-Arm Manipulation Through Collaborative Teleoperation","date":"2020-12-12","arxiv_id":"2012.06738","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-based-active-camera-control-with","title":"Imitation-Based Active Camera Control with Deep Convolutional Neural Network","date":"2020-12-11","arxiv_id":"2012.06428","repositories_listed":0,"syntology":null},{"url":null,"slug":"flatland-rl-multi-agent-reinforcement","title":"Flatland-RL : Multi-Agent Reinforcement Learning on Trains","date":"2020-12-10","arxiv_id":"2012.05893","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-rate-control-for-video-encoding-using","title":"Neural Rate Control for Video Encoding using Imitation Learning","date":"2020-12-09","arxiv_id":"2012.05339","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-eye-gaze-augmentation-to-enhance","title":"Selective Eye-gaze Augmentation To Enhance Imitation Learning In Atari Games","date":"2020-12-05","arxiv_id":"2012.03145","repositories_listed":0,"syntology":null},{"url":"/paper/neural-dynamic-policies-for-end-to-end-1","slug":"neural-dynamic-policies-for-end-to-end-1","title":"Neural Dynamic Policies for End-to-End Sensorimotor Learning","date":"2020-12-04","arxiv_id":"2012.02788","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-multi-type-mean-field-multi-agent","title":"Bayesian Multi-type Mean Field Multi-agent Imitation Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"f-gail-learning-f-divergence-for-generative-1","title":"f-GAIL: Learning f-Divergence for Generative Adversarial Imitation Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"milp-based-imitation-learning-for-hvac","title":"MILP-based Imitation Learning for HVAC control","date":"2020-12-01","arxiv_id":"2012.00286","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-imitation-learning-with-a","title":"Offline Imitation Learning with a Misspecified Simulator","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distilled-thompson-sampling-practical-and","title":"Distilled Thompson Sampling: Practical and Efficient Thompson Sampling via Imitation Learning","date":"2020-11-29","arxiv_id":"2011.14266","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-imitation-learning-for-real-time","title":"Hybrid Imitation Learning for Real-Time Service Restoration in Resilient Distribution Systems","date":"2020-11-29","arxiv_id":"2011.14458","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-agent-cooperation-in-bridge-bidding","title":"Human-Agent Cooperation in Bridge Bidding","date":"2020-11-28","arxiv_id":"2011.14124","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-learning-from-demonstrations-and","title":"Offline Learning from Demonstrations and Unlabeled Experience","date":"2020-11-27","arxiv_id":"2011.13885","repositories_listed":0,"syntology":null},{"url":null,"slug":"diluted-near-optimal-expert-demonstrations","title":"Diluted Near-Optimal Expert Demonstrations for Guiding Dialogue Stochastic Policy Optimisation","date":"2020-11-25","arxiv_id":"2012.04687","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-guided-navigation-via-cross-modal","title":"Language-guided Navigation via Cross-Modal Grounding and Alternate Adversarial Learning","date":"2020-11-22","arxiv_id":"2011.10972","repositories_listed":0,"syntology":null},{"url":null,"slug":"safari-safe-and-active-robot-imitation","title":"SAFARI: Safe and Active Robot Imitation Learning with Imagination","date":"2020-11-18","arxiv_id":"2011.09586","repositories_listed":0,"syntology":null},{"url":null,"slug":"grasping-with-chopsticks-combating-covariate","title":"Grasping with Chopsticks: Combating Covariate Shift in Model-free Imitation Learning for Fine Manipulation","date":"2020-11-13","arxiv_id":"2011.06719","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-generation-using-bilateral-control","title":"Motion Generation Using Bilateral Control-Based Imitation Learning with Autoregressive Learning","date":"2020-11-12","arxiv_id":"2011.06192","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-for-one-shot-visual-imitation","title":"Transformers for One-Shot Visual Imitation","date":"2020-11-11","arxiv_id":"2011.05970","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-trajectory-planning-using-reinforcement","title":"Safe Trajectory Planning Using Reinforcement Learning for Self Driving","date":"2020-11-09","arxiv_id":"2011.04702","repositories_listed":0,"syntology":null},{"url":null,"slug":"hilonet-hierarchical-imitation-learning-from","title":"HILONet: Hierarchical Imitation Learning from Non-Aligned Observations","date":"2020-11-05","arxiv_id":"2011.02671","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearl-non-explicit-action-reinforcement","title":"NEARL: Non-Explicit Action Reinforcement Learning for Robotic Control","date":"2020-11-02","arxiv_id":"2011.01046","repositories_listed":0,"syntology":null},{"url":null,"slug":"shaping-rewards-for-reinforcement-learning","title":"Shaping Rewards for Reinforcement Learning with Imperfect Demonstrations using Generative Models","date":"2020-11-02","arxiv_id":"2011.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"alibabas-submission-for-the-wmt-2020-ape","title":"Alibaba’s Submission for the WMT 2020 APE Shared Task: Improving Automatic Post-Editing with Pre-trained Conditional Cross-Lingual BERT","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pilot-efficient-planning-by-imitation","title":"PILOT: Efficient Planning by Imitation Learning and Optimisation for Safe Autonomous Driving","date":"2020-11-01","arxiv_id":"2011.00509","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-training-in-multi-agent","title":"Sample Efficient Training in Multi-Agent Adversarial Games with Limited Teammate Communication","date":"2020-11-01","arxiv_id":"2011.00424","repositories_listed":0,"syntology":null},{"url":null,"slug":"fighting-copycat-agents-in-behavioral-cloning","title":"Fighting Copycat Agents in Behavioral Cloning from Observation Histories","date":"2020-10-28","arxiv_id":"2010.14876","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-latent-movements-off-policy","title":"Contextual Latent-Movements Off-Policy Optimization for Robotic Manipulation Skills","date":"2020-10-26","arxiv_id":"2010.13766","repositories_listed":0,"syntology":null},{"url":null,"slug":"opal-offline-primitive-discovery-for-1","title":"OPAL: Offline Primitive Discovery for Accelerating Offline Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13611","repositories_listed":0,"syntology":null},{"url":null,"slug":"complex-skill-acquisition-through-simple-1","title":"Complex Skill Acquisition through Simple Skill Imitation Learning","date":"2020-10-23","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"error-bounds-of-imitating-policies-and","title":"Error Bounds of Imitating Policies and Environments","date":"2020-10-22","arxiv_id":"2010.11876","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-with-neural-density-models-1","title":"Imitation with Neural Density Models","date":"2020-10-19","arxiv_id":"2010.09808","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-select-nodes-in-bounded","title":"Learning to Select Nodes in Bounded Suboptimal Conflict-Based Search for Multi-Agent Path Finding","date":"2020-10-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-guaranteed-almost-equivalence-between","title":"On the Guaranteed Almost Equivalence between Imitation Learning from Observation and Demonstration","date":"2020-10-16","arxiv_id":"2010.08353","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-the-low-resource-challenge-for","title":"Tackling the Low-resource Challenge for Canonical Segmentation","date":"2020-10-06","arxiv_id":"2010.02804","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-dialogue-generation-by-imitating","title":"Regularizing Dialogue Generation by Imitating Implicit Scenarios","date":"2020-10-05","arxiv_id":"2010.01893","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-mixed","title":"Deep Reinforcement Learning with Mixed Convolutional Network","date":"2020-10-01","arxiv_id":"2010.00717","repositories_listed":0,"syntology":null},{"url":"/paper/multi-agent-social-reinforcement-learning","slug":"multi-agent-social-reinforcement-learning","title":"Emergent Social Learning via Multi-agent Reinforcement Learning","date":"2020-10-01","arxiv_id":"2010.00581","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-agent-social-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2010.00581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.00581"}},"official":null}},{"url":null,"slug":"population-guided-imitation-learning","title":"Population-Guided Imitation Learning","date":"2020-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"flight-connection-prediction-for-airline-crew","title":"Flight-connection Prediction for Airline Crew Scheduling to Construct Initial Clusters for OR Optimizer","date":"2020-09-26","arxiv_id":"2009.12501","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-in-deep-reinforcement","title":"Sim-to-Real Transfer in Deep Reinforcement Learning for Robotics: a Survey","date":"2020-09-24","arxiv_id":"2009.13303","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-the-reward-for-handwriting","title":"What is the Reward for Handwriting? -- Handwriting Generation by Imitation Learning","date":"2020-09-23","arxiv_id":"2009.10962","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-contraction-approach-to-model-based","title":"A Contraction Approach to Model-based Reinforcement Learning","date":"2020-09-18","arxiv_id":"2009.08586","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressed-imitation-learning","title":"Compressed imitation learning","date":"2020-09-18","arxiv_id":"2009.11697","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-selective-imitation","title":"Evolutionary Selective Imitation: Interpretable Agents by Imitation Learning Without a Demonstrator","date":"2020-09-17","arxiv_id":"2009.08403","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-the-fundamental-limits-of-imitation","title":"Toward the Fundamental Limits of Imitation Learning","date":"2020-09-13","arxiv_id":"2009.05990","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-for-neural-network","title":"Imitation Learning for Neural Network Autopilot in Fixed-Wing Unmanned Aerial Systems","date":"2020-09-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-by-observation-imitation-learning-for","title":"Learn by Observation: Imitation Learning for Drone Patrolling from Videos of A Human Navigator","date":"2020-08-30","arxiv_id":"2008.13193","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-based-lane-change","title":"Meta Reinforcement Learning-Based Lane Change Strategy for Autonomous Vehicles","date":"2020-08-28","arxiv_id":"2008.12451","repositories_listed":0,"syntology":null},{"url":null,"slug":"adail-adaptive-adversarial-imitation-learning","title":"ADAIL: Adaptive Adversarial Imitation Learning","date":"2020-08-23","arxiv_id":"2008.12647","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-adaptive-learning-for-runtime-resource","title":"Online Adaptive Learning for Runtime Resource Management of Heterogeneous SoCs","date":"2020-08-22","arxiv_id":"2008.09728","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-imitation-learning-via-random","title":"Adversarial Imitation Learning via Random Search","date":"2020-08-21","arxiv_id":"2008.09450","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-based-on-entropy","title":"Forward and inverse reinforcement learning sharing network weights and hyperparameters","date":"2020-08-17","arxiv_id":"2008.07284","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-imitation-made-easy","title":"Visual Imitation Made Easy","date":"2020-08-11","arxiv_id":"2008.04899","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-for-autonomous-trajectory","title":"Imitation Learning for Autonomous Trajectory Learning of Robot Arms in Space","date":"2020-08-10","arxiv_id":"2008.04007","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-based-dexterous-manipulations-with","title":"Physics-Based Dexterous Manipulations with Estimated Hand Poses and Residual Reinforcement Learning","date":"2020-08-07","arxiv_id":"2008.03285","repositories_listed":0,"syntology":null},{"url":null,"slug":"concurrent-training-improves-the-performance","title":"Concurrent Training Improves the Performance of Behavioral Cloning from Observation","date":"2020-08-03","arxiv_id":"2008.01205","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-the-race-between-deep-reinforcement","title":"Tracking the Race Between Deep Reinforcement Learning and Imitation Learning -- Extended Version","date":"2020-08-03","arxiv_id":"2008.00766","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-interactive-end-to-end-deep","title":"Sample Efficient Interactive End-to-End Deep Learning for Self-Driving Cars with Selective Multi-Class Safe Dataset Aggregation","date":"2020-07-29","arxiv_id":"2007.14671","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-imitation-gap-by-adaptive","title":"Bridging the Imitation Gap by Adaptive Insubordination","date":"2020-07-23","arxiv_id":"2007.12173","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-metrics-for-behaviour-modeling","title":"Evaluation metrics for behaviour modeling","date":"2020-07-23","arxiv_id":"2007.12298","repositories_listed":0,"syntology":null},{"url":null,"slug":"complex-skill-acquisition-through-simple","title":"Complex Skill Acquisition Through Simple Skill Imitation Learning","date":"2020-07-20","arxiv_id":"2007.10281","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-graphical-planner-contextual-global","title":"Evolving Graphical Planner: Contextual Global Planning for Vision-and-Language Navigation","date":"2020-07-11","arxiv_id":"2007.05655","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-autonomous-vehicle-control-in-the","title":"A Survey on Autonomous Vehicle Control in the Era of Mixed-Autonomy: From Physics-Based to AI-Guided Driving Policy Learning","date":"2020-07-10","arxiv_id":"2007.05156","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-an-automated-gesture-imitation-game","title":"Building an Automated Gesture Imitation Game for Teenagers with ASD","date":"2020-07-09","arxiv_id":"2007.04604","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-efficient-search-approximation-in","title":"A Study of Learning Search Approximation in Mixed Integer Branch and Bound: Node Selection in SCIP","date":"2020-07-08","arxiv_id":"2007.03948","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-approach-for-ai-driving","title":"Imitation Learning Approach for AI Driving Olympics Trained on Real-world and Simulation Data Simultaneously","date":"2020-07-07","arxiv_id":"2007.03514","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-fast-improvement-in-online-policy","title":"Explaining Fast Improvement in Online Imitation Learning","date":"2020-07-06","arxiv_id":"2007.02520","repositories_listed":0,"syntology":null},{"url":null,"slug":"cluzh-at-sigmorphon-2020-shared-task-on","title":"CLUZH at SIGMORPHON 2020 Shared Task on Multilingual Grapheme-to-Phoneme Conversion","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-action-dialog-policy-learning-with","title":"Multi-Action Dialog Policy Learning with Interactive Human Teaching","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-improvement-from-multiple-experts","title":"Policy Improvement via Imitation of Multiple Oracles","date":"2020-07-01","arxiv_id":"2007.00795","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-will-generative-adversarial-imitation","title":"When Will Generative Adversarial Imitation Learning Algorithms Attain Global Convergence","date":"2020-06-24","arxiv_id":"2006.13506","repositories_listed":0,"syntology":null},{"url":null,"slug":"pico-primitive-imitation-for-control","title":"PICO: Primitive Imitation for COntrol","date":"2020-06-22","arxiv_id":"2006.12551","repositories_listed":0,"syntology":null},{"url":null,"slug":"autood-automated-outlier-detection-via","title":"AutoOD: Automated Outlier Detection via Curiosity-guided Search and Self-imitation Learning","date":"2020-06-19","arxiv_id":"2006.11321","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-agent-policies-with-interpretable","title":"Modelling Agent Policies with Interpretable Imitation Learning","date":"2020-06-19","arxiv_id":"2006.11309","repositories_listed":0,"syntology":null},{"url":null,"slug":"reparameterized-variational-divergence-1","title":"Reparameterized Variational Divergence Minimization for Stable Imitation","date":"2020-06-18","arxiv_id":"2006.10810","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-imitation-learning-from-multiple-non","title":"Active Imitation Learning from Multiple Non-Deterministic Teachers: Formulation, Challenges, and Algorithms","date":"2020-06-14","arxiv_id":"2006.07777","repositories_listed":0,"syntology":null},{"url":"/paper/self-imitation-learning-via-generalized-lower","slug":"self-imitation-learning-via-generalized-lower","title":"Self-Imitation Learning via Generalized Lower Bound Q-learning","date":"2020-06-12","arxiv_id":"2006.07442","repositories_listed":0,"syntology":{"n":20,"n_ran":16,"n_constructed":4,"n_ran_checked":13,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":2,"n_no_contract":11,"n_pointer_only":11,"phrase":"16 ran (of which 4 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 2 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-imitation-learning-via-generalized-lower#ran","syntology_url":"https://syntology.ai/paper/2006.07442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07442"}},"official":null}},{"url":null,"slug":"pac-bounds-for-imitation-and-model-based","title":"PAC Bounds for Imitation and Model-based Batch Learning of Contextual Markov Decision Processes","date":"2020-06-11","arxiv_id":"2006.06352","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-human-driving-behavior-through","title":"Modeling Human Driving Behavior through Generative Adversarial Imitation Learning","date":"2020-06-10","arxiv_id":"2006.06412","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealing-deep-reinforcement-learning-models","title":"Stealing Deep Reinforcement Learning Models for Fun and Profit","date":"2020-06-09","arxiv_id":"2006.05032","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-autonomous-driving-by-learning-end","title":"Explaining Autonomous Driving by Learning End-to-End Visual Attention","date":"2020-06-05","arxiv_id":"2006.03347","repositories_listed":0,"syntology":null}],"record_sha256":"255d6c95ea8996d8d65cd6f5d0596a21fe15a28c644fbb3ca19671c0744f7a2f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}