{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/imitation-learning/papers/12","list_of":"/task/imitation-learning","task":"Imitation Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":12,"pages_in_order":22,"rows_per_page":100,"rows":[1101,1200],"of":2122,"counts":{"archive_papers_tagged":2122,"with_a_code_link":691,"where_syntology_ran_a_sample":234,"not_listed_spam_title":0,"listed":2122,"listed_where_code_ran":234,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":191,"every_run_a_failure_of_syntologys_instrument":43,"listed_with_a_run_with_no_instrument_failure":191,"listed_every_run_a_failure_of_syntologys_instrument":43,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/imitation-learning","prev":"/task/imitation-learning/papers/11","next":"/task/imitation-learning/papers/13","papers":[{"url":null,"slug":"interpretable-modeling-of-deep-reinforcement","title":"Interpretable Modeling of Deep Reinforcement Learning Driven Scheduling","date":"2024-03-24","arxiv_id":"2403.16293","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-feature-selection-for-inverse","title":"Automated Feature Selection for Inverse Reinforcement Learning","date":"2024-03-22","arxiv_id":"2403.15079","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmented-reality-demonstrations-for-scalable","title":"Augmented Reality Demonstrations for Scalable Robot Imitation Learning","date":"2024-03-20","arxiv_id":"2403.13910","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-distillation-for","title":"Information-Theoretic Distillation for Reference-less Summarization","date":"2024-03-20","arxiv_id":"2403.13780","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-visual-imitation-learning-for","title":"Adaptive Visual Imitation Learning for Robotic Assisted Feeding Across Varied Bowl Configurations and Food Types","date":"2024-03-19","arxiv_id":"2403.12891","repositories_listed":0,"syntology":null},{"url":null,"slug":"anyskill-learning-open-vocabulary-physical","title":"AnySkill: Learning Open-Vocabulary Physical Skill for Interactive Agents","date":"2024-03-19","arxiv_id":"2403.12835","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-ais-are-not-learning-and-why-bio","title":"What AIs are not Learning (and Why)","date":"2024-03-19","arxiv_id":"2404.04267","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-reinforcement-learning-with","title":"Bootstrapping Reinforcement Learning with Imitation for Vision-Based Agile Flight","date":"2024-03-18","arxiv_id":"2403.12203","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-fine-tuning-as-inverse","title":"Supervised Fine-Tuning as Inverse Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.12017","repositories_listed":0,"syntology":null},{"url":null,"slug":"visuo-tactile-pretraining-for-cable-plugging","title":"VITaL Pretraining: Visuo-Tactile Pretraining for Tactile and Non-Tactile Manipulation Policies","date":"2024-03-18","arxiv_id":"2403.11898","repositories_listed":0,"syntology":null},{"url":null,"slug":"sculptdiff-learning-robotic-clay-sculpting","title":"SculptDiff: Learning Robotic Clay Sculpting from Humans with Goal Conditioned Diffusion Policy","date":"2024-03-15","arxiv_id":"2403.10401","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexcap-scalable-and-portable-mocap-data","title":"DexCap: Scalable and Portable Mocap Data Collection System for Dexterous Manipulation","date":"2024-03-12","arxiv_id":"2403.07788","repositories_listed":0,"syntology":null},{"url":null,"slug":"telemoma-a-modular-and-versatile","title":"TeleMoMa: A Modular and Versatile Teleoperation System for Mobile Manipulation","date":"2024-03-12","arxiv_id":"2403.07869","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-neural-motion-planning-on","title":"Physics-informed Neural Motion Planning on Constraint Manifolds","date":"2024-03-09","arxiv_id":"2403.05765","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-data-collection-for-robotic","title":"Efficient Data Collection for Robotic Manipulation via Compositional Generalization","date":"2024-03-08","arxiv_id":"2403.05110","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconciling-reality-through-simulation-a-real","title":"Reconciling Reality through Simulation: A Real-to-Sim-to-Real Approach for Robust Manipulation","date":"2024-03-06","arxiv_id":"2403.03949","repositories_listed":0,"syntology":null},{"url":null,"slug":"rt-h-action-hierarchies-using-language","title":"RT-H: Action Hierarchies Using Language","date":"2024-03-04","arxiv_id":"2403.01823","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-mean-zero-disagreement-regularized","title":"Continuous Mean-Zero Disagreement-Regularized Imitation Learning (CMZ-DRIL)","date":"2024-03-02","arxiv_id":"2403.01059","repositories_listed":0,"syntology":null},{"url":null,"slug":"prime-scaffolding-manipulation-tasks-with","title":"PRIME: Scaffolding Manipulation Tasks with Behavior Primitives for Data-Efficient Imitation Learning","date":"2024-03-01","arxiv_id":"2403.00929","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-policy-learning-via-offline-skill","title":"Robust Policy Learning via Offline Skill Diffusion","date":"2024-03-01","arxiv_id":"2403.00225","repositories_listed":0,"syntology":null},{"url":null,"slug":"digic-domain-generalizable-imitation-learning","title":"DIGIC: Domain Generalizable Imitation Learning by Causal Discovery","date":"2024-02-29","arxiv_id":"2402.18910","repositories_listed":0,"syntology":null},{"url":null,"slug":"ela-exploited-level-augmentation-for-offline","title":"ELA: Exploited Level Augmentation for Offline Learning in Zero-Sum Games","date":"2024-02-28","arxiv_id":"2402.18617","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-regularized-optimal-transport-on","title":"Imitation-regularized Optimal Transport on Networks: Provable Robustness and Application to Logistics Planning","date":"2024-02-28","arxiv_id":"2402.17967","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-language-guided-state","title":"Learning with Language-Guided State Abstractions","date":"2024-02-28","arxiv_id":"2402.18759","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-meets-dagger-supercharging-eye-in","title":"Diffusion Meets DAgger: Supercharging Eye-in-hand Imitation Learning","date":"2024-02-27","arxiv_id":"2402.17768","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-mutual-information-for-language","title":"Rethinking Mutual Information for Language Conditioned Skill Discovery on Imitation Learning","date":"2024-02-27","arxiv_id":"2402.17511","repositories_listed":0,"syntology":null},{"url":null,"slug":"c-gail-stabilizing-generative-adversarial","title":"C-GAIL: Stabilizing Generative Adversarial Imitation Learning with Control Theory","date":"2024-02-26","arxiv_id":"2402.16349","repositories_listed":0,"syntology":null},{"url":null,"slug":"expressive-whole-body-control-for-humanoid","title":"Expressive Whole-Body Control for Humanoid Robots","date":"2024-02-26","arxiv_id":"2402.16796","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-translations-emergent-communication","title":"Learning Translations: Emergent Communication Pretraining for Cooperative Language Acquisition","date":"2024-02-26","arxiv_id":"2402.16247","repositories_listed":0,"syntology":null},{"url":null,"slug":"betail-behavior-transformer-adversarial","title":"BeTAIL: Behavior Transformer Adversarial Imitation Learning from Human Racing Gameplay","date":"2024-02-22","arxiv_id":"2402.14194","repositories_listed":0,"syntology":null},{"url":null,"slug":"cyberdemo-augmenting-simulated-human","title":"CyberDemo: Augmenting Simulated Human Demonstration for Real-World Dexterous Manipulation","date":"2024-02-22","arxiv_id":"2402.14795","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-planning-based-on-2d-object-bounding-box","title":"Path Planning based on 2D Object Bounding-box","date":"2024-02-22","arxiv_id":"2402.14933","repositories_listed":0,"syntology":null},{"url":null,"slug":"align-your-intents-offline-imitation-learning","title":"Align Your Intents: Offline Imitation Learning via Optimal Transport","date":"2024-02-20","arxiv_id":"2402.13037","repositories_listed":0,"syntology":null},{"url":null,"slug":"dinobot-robot-manipulation-via-retrieval-and","title":"DINOBot: Robot Manipulation via Retrieval and Alignment with Vision Foundation Models","date":"2024-02-20","arxiv_id":"2402.13181","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-generative-adversarial","title":"Interpretable Generative Adversarial Imitation Learning","date":"2024-02-15","arxiv_id":"2402.10310","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-reset-divide-conquer-imitation","title":"Single-Reset Divide & Conquer Imitation Learning","date":"2024-02-14","arxiv_id":"2402.09355","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-driven-imitation-of-subrational-behavior","title":"LLM-driven Imitation of Subrational Behavior : Illusion or Reality?","date":"2024-02-13","arxiv_id":"2402.08755","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-imitation-in-a-non-stationary","title":"One-shot Imitation in a Non-Stationary Environment via Multi-Modal Skill","date":"2024-02-13","arxiv_id":"2402.08369","repositories_listed":0,"syntology":null},{"url":null,"slug":"cambranch-contrastive-learning-with-augmented","title":"CAMBranch: Contrastive Learning with Augmented MILPs for Branching","date":"2024-02-06","arxiv_id":"2402.03647","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-inverse-reinforcement-learning","title":"Accelerating Inverse Reinforcement Learning with Expert Bootstrapping","date":"2024-02-04","arxiv_id":"2402.02608","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffstitch-boosting-offline-reinforcement","title":"DiffStitch: Boosting Offline Reinforcement Learning with Diffusion-based Trajectory Stitching","date":"2024-02-04","arxiv_id":"2402.02439","repositories_listed":0,"syntology":null},{"url":null,"slug":"extrinsicaly-rewarded-soft-q-imitation","title":"Extrinsicaly Rewarded Soft Q Imitation Learning with Discriminator","date":"2024-01-30","arxiv_id":"2401.16772","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-former-stitching-via-latent","title":"Context-Former: Stitching via Latent Conditioned Sequence Modeling","date":"2024-01-29","arxiv_id":"2401.16452","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-cooperation-under-punishment-a","title":"Emergence of cooperation under punishment: A reinforcement learning perspective","date":"2024-01-29","arxiv_id":"2401.16073","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-imitation-policy-via-search-in","title":"Zero-shot Imitation Policy via Search in Demonstration Dataset","date":"2024-01-29","arxiv_id":"2401.16398","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-object-navigation-in-real-environments","title":"Multi-Object Navigation in real environments using hybrid policies","date":"2024-01-24","arxiv_id":"2401.13800","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-generalized-end-to-end-autonomous","title":"Efficient and Generalized end-to-end Autonomous Driving System with Latent Deep Reinforcement Learning and Demonstrations","date":"2024-01-22","arxiv_id":"2401.11792","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-generative-adversarial","title":"Multi-Agent Generative Adversarial Interactive Self-Imitation Learning for AUV Formation Control and Obstacle Avoidance","date":"2024-01-21","arxiv_id":"2401.11378","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-imitation-learning-with-calibrated","title":"Visual Imitation Learning with Calibrated Contrastive Representation","date":"2024-01-21","arxiv_id":"2401.11396","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-inputting-image-feature-to","title":"Imitation Learning Inputting Image Feature to Each Layer of Neural Network","date":"2024-01-18","arxiv_id":"2401.09691","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-imitation-learning-by-controlling-the","title":"Offline Imitation Learning by Controlling the Effective Planning Horizon","date":"2024-01-18","arxiv_id":"2401.09728","repositories_listed":0,"syntology":null},{"url":null,"slug":"swbt-similarity-weighted-behavior-transformer","title":"Learning from Imperfect Demonstrations with Self-Supervision for Robotic Manipulation","date":"2024-01-17","arxiv_id":"2401.08957","repositories_listed":0,"syntology":null},{"url":null,"slug":"agentmixer-multi-agent-correlated-policy","title":"AgentMixer: Multi-Agent Correlated Policy Factorization","date":"2024-01-16","arxiv_id":"2401.08728","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-stable-koopman-embeddings-for","title":"Learning Stable Koopman Embeddings for Identification and Control","date":"2024-01-16","arxiv_id":"2401.08153","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-imitation-of-human-actions","title":"Robotic Imitation of Human Actions","date":"2024-01-16","arxiv_id":"2401.08381","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-robot-data-for-dual-arm-fine","title":"Multi-task real-robot data with gaze attention for dual-arm fine manipulation","date":"2024-01-15","arxiv_id":"2401.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"coin-chance-constrained-imitation-learning","title":"COIN: Chance-Constrained Imitation Learning for Uncertainty-aware Adaptive Resource Oversubscription Policy","date":"2024-01-13","arxiv_id":"2401.07051","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-imitation-learning-for-automated-game","title":"Robust Imitation Learning for Automated Game Testing","date":"2024-01-09","arxiv_id":"2401.04572","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioural-cloning-in-vizdoom","title":"Behavioural Cloning in VizDoom","date":"2024-01-08","arxiv_id":"2401.03993","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-navigation-in-complex-environments-1","title":"Autonomous Navigation in Complex Environments","date":"2024-01-06","arxiv_id":"2401.03267","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobile-aloha-learning-bimanual-mobile","title":"Mobile ALOHA: Learning Bimanual Mobile Manipulation with Low-Cost Whole-Body Teleoperation","date":"2024-01-04","arxiv_id":"2401.02117","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-adaptable-and-generalizable","title":"Towards an Adaptable and Generalizable Optimization Engine in Decision and Control: A Meta Reinforcement Learning Approach","date":"2024-01-04","arxiv_id":"2401.02508","repositories_listed":0,"syntology":null},{"url":null,"slug":"genh2r-learning-generalizable-human-to-robot","title":"GenH2R: Learning Generalizable Human-to-Robot Handover via Scalable Simulation, Demonstration, and Imitation","date":"2024-01-01","arxiv_id":"2401.00929","repositories_listed":0,"syntology":null},{"url":null,"slug":"genh2r-learning-generalizable-human-to-robot-1","title":"GenH2R: Learning Generalizable Human-to-Robot Handover via Scalable Simulation Demonstration and Imitation","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"active-third-person-imitation-learning","title":"Active Third-Person Imitation Learning","date":"2023-12-27","arxiv_id":"2312.16365","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-integrated-imitation-and-reinforcement","title":"An Integrated Imitation and Reinforcement Learning Methodology for Robust Agile Aircraft Control with Limited Pilot Demonstration Data","date":"2023-12-27","arxiv_id":"2401.08663","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-retrieval-alignment","title":"On the Effectiveness of Retrieval, Alignment, and Replay in Manipulation","date":"2023-12-19","arxiv_id":"2312.12345","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-relay-learning-optimization-approach","title":"Stable Relay Learning Optimization Approach for Fast Power System Production Cost Minimization Simulation","date":"2023-12-19","arxiv_id":"2312.11896","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-gradient-explosion-in-generative","title":"Exploring Gradient Explosion in Generative Adversarial Imitation Learning: A Probabilistic Perspective","date":"2023-12-18","arxiv_id":"2312.11214","repositories_listed":0,"syntology":null},{"url":null,"slug":"movement-primitive-diffusion-learning-gentle","title":"Movement Primitive Diffusion: Learning Gentle Robotic Manipulation of Deformable Objects","date":"2023-12-15","arxiv_id":"2312.10008","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-differentiable-integral-control","title":"Neural Differentiable Integral Control Barrier Functions for Unknown Nonlinear Systems with Input Constraints","date":"2023-12-12","arxiv_id":"2312.07345","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-end-to-end-reinforcement-learning-for","title":"Partial End-to-end Reinforcement Learning for Robustness Against Modelling Error in Autonomous Racing","date":"2023-12-11","arxiv_id":"2312.06406","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-based-prediction-and-planning-policy","title":"Graph-based Prediction and Planning Policy Network (GP3Net) for scalable self-driving in dynamic environments using Deep Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05784","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-conditioned-semantic-search-based","title":"Language-Conditioned Semantic Search-Based Policy for Robotic Manipulation Tasks","date":"2023-12-10","arxiv_id":"2312.05925","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-representations-pretrained-with","title":"Understanding Representations Pretrained with Auxiliary Losses for Embodied Agent Planning","date":"2023-12-06","arxiv_id":"2312.10069","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitating-shortest-paths-in-simulation","title":"SPOC: Imitating Shortest Paths in Simulation Enables Effective Navigation and Manipulation in the Real World","date":"2023-12-05","arxiv_id":"2312.02976","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-hindsight-self-imitation-learning-for","title":"Visual Hindsight Self-Imitation Learning for Interactive Navigation","date":"2023-12-05","arxiv_id":"2312.03446","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-encoders-for-data-efficient-imitation","title":"Visual Encoders for Data-Efficient Imitation Learning in Modern Video Games","date":"2023-12-04","arxiv_id":"2312.02312","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptive-imitation-learning-with-1","title":"Domain Adaptive Imitation Learning with Visual Observation","date":"2023-12-01","arxiv_id":"2312.00548","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-based-concave-utility","title":"Efficient Model-Based Concave Utility Reinforcement Learning through Greedy Mirror Descent","date":"2023-11-30","arxiv_id":"2311.18346","repositories_listed":0,"syntology":null},{"url":null,"slug":"md-splatting-learning-metric-deformation-from","title":"DeformGS: Scene Flow in Highly Deformable Scenes for Deformable Object Manipulation","date":"2023-11-30","arxiv_id":"2312.00583","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-in-robotics-an-upcoming","title":"Transfer Learning in Robotics: An Upcoming Breakthrough? A Review of Promises and Challenges","date":"2023-11-29","arxiv_id":"2311.18044","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-power-flow-in-highly-renewable-power","title":"Optimal Power Flow in Highly Renewable Power System Based on Attention Neural Networks","date":"2023-11-23","arxiv_id":"2311.13949","repositories_listed":0,"syntology":null},{"url":null,"slug":"tube-nerf-efficient-imitation-learning-of","title":"Tube-NeRF: Efficient Imitation Learning of Visuomotor Policies from MPC using Tube-Guided Data Augmentation and NeRFs","date":"2023-11-23","arxiv_id":"2311.14153","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-and-imitation-learning","title":"Curriculum Learning and Imitation Learning for Model-free Control on Financial Time-series","date":"2023-11-22","arxiv_id":"2311.13326","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlif-interactive-imitation-learning-as","title":"RLIF: Interactive Imitation Learning as Reinforcement Learning","date":"2023-11-21","arxiv_id":"2311.12996","repositories_listed":0,"syntology":null},{"url":"/paper/orca-2-teaching-small-language-models-how-to","slug":"orca-2-teaching-small-language-models-how-to","title":"Orca 2: Teaching Small Language Models How to Reason","date":"2023-11-18","arxiv_id":"2311.11045","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalizable-imitation-learning-through-pre","title":"Generalizable Imitation Learning Through Pre-Trained Representations","date":"2023-11-15","arxiv_id":"2311.09350","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-imitation-learning-on-aggregated","title":"Adversarial Imitation Learning On Aggregated Data","date":"2023-11-14","arxiv_id":"2311.08568","repositories_listed":0,"syntology":null},{"url":null,"slug":"extending-multilingual-machine-translation","title":"Extending Multilingual Machine Translation through Imitation Learning","date":"2023-11-14","arxiv_id":"2311.08538","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncommonsense-reasoning-abductive-reasoning","title":"UNcommonsense Reasoning: Abductive Reasoning about Uncommon Situations","date":"2023-11-14","arxiv_id":"2311.08469","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-motion-prediction-with-cognitive","title":"Social Motion Prediction with Cognitive Hierarchies","date":"2023-11-08","arxiv_id":"2311.04726","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-efficient-reinforcement-learning-with","title":"Time-Efficient Reinforcement Learning with Stochastic Stateful Policies","date":"2023-11-07","arxiv_id":"2311.04082","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-learning-based-alternative-multi","title":"Imitation Learning based Alternative Multi-Agent Proximal Policy Optimization for Well-Formed Swarm-Oriented Pursuit Avoidance","date":"2023-11-06","arxiv_id":"2311.02912","repositories_listed":0,"syntology":null},{"url":null,"slug":"maaip-multi-agent-adversarial-interaction","title":"MAAIP: Multi-Agent Adversarial Interaction Priors for imitation from fighting demonstrations for physics-based characters","date":"2023-11-04","arxiv_id":"2311.02502","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-bootstrapped-reinforcement-learning","title":"Imitation Bootstrapped Reinforcement Learning","date":"2023-11-03","arxiv_id":"2311.02198","repositories_listed":0,"syntology":null},{"url":null,"slug":"lotus-continual-imitation-learning-for-robot","title":"LOTUS: Continual Imitation Learning for Robot Manipulation Through Unsupervised Skill Discovery","date":"2023-11-03","arxiv_id":"2311.02058","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-statistical-guarantee-for-representation","title":"A Statistical Guarantee for Representation Transfer in Multitask Imitation Learning","date":"2023-11-02","arxiv_id":"2311.01589","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-realistic-traffic-agents-in-closed","title":"Learning Realistic Traffic Agents in Closed-loop","date":"2023-11-02","arxiv_id":"2311.01394","repositories_listed":0,"syntology":null},{"url":"/paper/vision-language-foundation-models-as","slug":"vision-language-foundation-models-as","title":"Vision-Language Foundation Models as Effective Robot Imitators","date":"2023-11-02","arxiv_id":"2311.01378","repositories_listed":0,"syntology":null}],"record_sha256":"cd8ff53ff3a51bf9f943632d2d6e0790c0008a2c3dbc358cea30f84e7f7e26de","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}