{"url":"/task/chatbot","name":"Chatbot","slug":"chatbot","description_markdown":"**Chatbot** or conversational AI is a language model designed and implemented to have conversations with humans. \r\n\r\n\r\n<span class=\"description-source\">Source: [Open Data Chatbot ](https://arxiv.org/abs/1909.03653)</span>\r\n\r\n[Image source](https://arxiv.org/pdf/2006.16779v3.pdf)","categories":[{"name":"Methodology","url":"/area/methodology"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":971,"papers_with_code":269,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":10,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/chatbot-on-alpacaeval","slug":"chatbot-on-alpacaeval","dataset":"AlpacaEval","dataset_url":"/dataset/alpacaeval","rows_in_archive":1,"metrics":["Average win rate"],"first_row_in_archive_order":{"model":"Yi 34B Chat","paper_title":"Yi: Open Foundation Models by 01.AI","paper_url":"/paper/yi-open-foundation-models-by-01-ai","paper_date":"2024-03-07","arxiv_id":"2403.04652","code_links":[{"title":"01-ai/yi","url":"https://github.com/01-ai/yi"}],"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/alpacaeval","name":"AlpacaEval","full_name":"","num_papers_in_archive":131},{"url":"/dataset/blended-skill-talk","name":"Blended Skill Talk","full_name":"Blended Skill Talk","num_papers_in_archive":32},{"url":"/dataset/photobook","name":"PhotoBook","full_name":"","num_papers_in_archive":11},{"url":"/dataset/taiga-corpus","name":"Taiga Corpus","full_name":"An open-source corpus for machine learning.","num_papers_in_archive":5},{"url":"/dataset/metaphorical-connections","name":"Metaphorical Connections","full_name":null,"num_papers_in_archive":2},{"url":"/dataset/alpacaeval-th","name":"AlpacaEval-TH","full_name":"","num_papers_in_archive":1},{"url":"/dataset/chatgpt-software-testing-study","name":"ChatGPT Software Testing Study","full_name":"ChatGPT Software Testing Study Dataset","num_papers_in_archive":1},{"url":"/dataset/finchat","name":"FinChat","full_name":"Finnish Chat Conversations on Everyday Topics","num_papers_in_archive":1},{"url":"/dataset/mt-bench-th","name":"MT-Bench-TH","full_name":"","num_papers_in_archive":1},{"url":"/dataset/chatgpt-software-testing","name":"ChatGPT-software-testing","full_name":"ChatGPT Software Testing","num_papers_in_archive":0}],"subtasks":[{"url":"/task/dialogue-generation","name":"Dialogue Generation"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":269,"tagged_in_all":971,"items":[{"url":"/paper/qlora-efficient-finetuning-of-quantized-llms","title":"QLoRA: Efficient Finetuning of Quantized LLMs","date":"2023-05-23","arxiv_id":"2305.14314","repositories_listed":20,"syntology":{"n":26,"n_ran":17,"n_unverified":9,"n_pointer_only":17}},{"url":"/paper/end-to-end-task-completion-neural-dialogue","title":"End-to-End Task-Completion Neural Dialogue Systems","date":"2017-03-03","arxiv_id":"1703.01008","repositories_listed":13,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":5}},{"url":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","date":"2023-06-09","arxiv_id":"2306.05685","repositories_listed":11,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/visual-dialog","title":"Visual Dialog","date":"2016-11-26","arxiv_id":"1611.08669","repositories_listed":11,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/recipes-for-building-an-open-domain-chatbot","title":"Recipes for building an open-domain chatbot","date":"2020-04-28","arxiv_id":"2004.13637","repositories_listed":8,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-dialogue","title":"Deep Reinforcement Learning for Dialogue Generation","date":"2016-06-05","arxiv_id":"1606.01541","repositories_listed":8,"syntology":null},{"url":"/paper/mistral-7b","title":"Mistral 7B","date":"2023-10-10","arxiv_id":"2310.06825","repositories_listed":6,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":1}},{"url":"/paper/chatbot-arena-an-open-platform-for-evaluating","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","date":"2024-03-07","arxiv_id":"2403.04132","repositories_listed":5,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/lmsys-chat-1m-a-large-scale-real-world-llm","title":"LMSYS-Chat-1M: A Large-Scale Real-World LLM Conversation Dataset","date":"2023-09-21","arxiv_id":"2309.11998","repositories_listed":5,"syntology":null},{"url":"/paper/baize-an-open-source-chat-model-with","title":"Baize: An Open-Source Chat Model with Parameter-Efficient Tuning on Self-Chat Data","date":"2023-04-03","arxiv_id":"2304.01196","repositories_listed":5,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/from-crowdsourced-data-to-high-quality","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","date":"2024-06-17","arxiv_id":"2406.11939","repositories_listed":4,"syntology":{"n":21,"n_ran":18,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/wildteaming-at-scale-from-in-the-wild","title":"WildTeaming at Scale: From In-the-Wild Jailbreaks to (Adversarially) Safer Language Models","date":"2024-06-26","arxiv_id":"2406.18510","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/rlhf-workflow-from-reward-modeling-to-online","title":"RLHF Workflow: From Reward Modeling to Online RLHF","date":"2024-05-13","arxiv_id":"2405.07863","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/don-t-forget-your-abc-s-evaluating-the-state","title":"Don't Forget Your ABC's: Evaluating the State-of-the-Art in Chat-Oriented Dialogue Systems","date":"2022-12-18","arxiv_id":"2212.09180","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/plato-2-towards-building-an-open-domain","title":"PLATO-2: Towards Building an Open-Domain Chatbot via Curriculum Learning","date":"2020-06-30","arxiv_id":"2006.16779","repositories_listed":3,"syntology":null},{"url":"/paper/subword-semantic-hashing-for-intent","title":"Subword Semantic Hashing for Intent Classification on Small Datasets","date":"2018-10-16","arxiv_id":"1810.07150","repositories_listed":3,"syntology":null},{"url":"/paper/llama-omni2-llm-based-real-time-spoken","title":"LLaMA-Omni2: LLM-based Real-time Spoken Chatbot with Autoregressive Streaming Speech Synthesis","date":"2025-05-05","arxiv_id":"2505.02625","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_unverified":3,"n_pointer_only":10}},{"url":"/paper/eliza-reanimated-the-world-s-first-chatbot","title":"ELIZA Reanimated: The world's first chatbot restored on the world's first time sharing system","date":"2025-01-12","arxiv_id":"2501.06707","repositories_listed":2,"syntology":null},{"url":"/paper/low-code-from-frontend-to-backend-connecting","title":"Low-code from frontend to backend: Connecting conversational user interfaces to backend services via a low-code IoT platform","date":"2024-09-13","arxiv_id":"2410.00006","repositories_listed":2,"syntology":null},{"url":"/paper/jamba-1-5-hybrid-transformer-mamba-models-at","title":"Jamba-1.5: Hybrid Transformer-Mamba Models at Scale","date":"2024-08-22","arxiv_id":"2408.12570","repositories_listed":2,"syntology":null},{"url":"/paper/simpo-simple-preference-optimization-with-a","title":"SimPO: Simple Preference Optimization with a Reference-Free Reward","date":"2024-05-23","arxiv_id":"2405.14734","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/using-adaptive-empathetic-responses-for","title":"Using Adaptive Empathetic Responses for Teaching English","date":"2024-04-21","arxiv_id":"2404.13764","repositories_listed":2,"syntology":null},{"url":"/paper/length-controlled-alpacaeval-a-simple-way-to","title":"Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators","date":"2024-04-06","arxiv_id":"2404.04475","repositories_listed":2,"syntology":null},{"url":"/paper/three-ways-of-using-large-language-models-to","title":"Three Ways of Using Large Language Models to Evaluate Chat","date":"2023-08-12","arxiv_id":"2308.06502","repositories_listed":2,"syntology":null},{"url":"/paper/educhat-a-large-scale-language-model-based","title":"EduChat: A Large-Scale Language Model-based Chatbot System for Intelligent Education","date":"2023-08-05","arxiv_id":"2308.02773","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_unverified":2,"n_pointer_only":9}},{"url":"/paper/h2ogpt-democratizing-large-language-models","title":"h2oGPT: Democratizing Large Language Models","date":"2023-06-13","arxiv_id":"2306.08161","repositories_listed":2,"syntology":null},{"url":"/paper/task-optimized-adapters-for-an-end-to-end","title":"Task-Optimized Adapters for an End-to-End Task-Oriented Dialogue System","date":"2023-05-04","arxiv_id":"2305.02468","repositories_listed":2,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/chai-a-chatbot-ai-for-task-oriented-dialogue","title":"CHAI: A CHatbot AI for Task-Oriented Dialogue with Offline Reinforcement Learning","date":"2022-04-18","arxiv_id":"2204.08426","repositories_listed":2,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/eva2-0-investigating-open-domain-chinese","title":"EVA2.0: Investigating Open-Domain Chinese Dialogue Systems with Large-Scale Pre-Training","date":"2022-03-17","arxiv_id":"2203.09313","repositories_listed":2,"syntology":{"n":13,"n_ran":2,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/automatic-evaluation-and-moderation-of-open","title":"Automatic Evaluation and Moderation of Open-domain Dialogue Systems","date":"2021-11-03","arxiv_id":"2111.02110","repositories_listed":2,"syntology":null}],"syntology_records":17,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}