{"url":"/dataset/irc-disentanglement","name":"irc-disentanglement","full_name":null,"description_markdown":"This is a dataset for disentangling conversations on IRC, which is the task of identifying separate conversations in a single stream of messages. It contains disentanglement information for 77,563 messages or IRC.\n\nSource: [https://github.com/jkkummerfeld/irc-disentanglement](https://github.com/jkkummerfeld/irc-disentanglement)\nImage Source: [https://github.com/jkkummerfeld/irc-disentanglement](https://github.com/jkkummerfeld/irc-disentanglement)","description_withheld":null,"homepage":"https://github.com/jkkummerfeld/irc-disentanglement","introduced_date":null,"introduced_date_note":null,"introduced_by":{"paper":"/paper/analyzing-assumptions-in-conversation","title":"A Large-Scale Corpus for Conversation Disentanglement","first_author":"Jonathan K. Kummerfeld","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Language Modelling","url":"/task/language-modelling","datasets_with_task":"/datasets/task/language-modelling"},{"name":"Conversation Disentanglement","url":"/task/conversation-disentanglement","datasets_with_task":"/datasets/task/conversation-disentanglement"}],"languages":[],"variants":["irc-disentanglement","Linux IRC (Ch2 Kummerfeld)","Linux IRC (Ch2 Elsner)"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/jkkummerfeld/irc_disentangle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/irc_disentangle","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tensorflow/datasets","url":"https://www.tensorflow.org/datasets/catalog/irc_disentanglement","frameworks":["tf","jax"]},{"repo":"https://github.com/jkkummerfeld/irc-disentanglement","url":"https://github.com/jkkummerfeld/irc-disentanglement","frameworks":[]}],"num_papers_in_archive":4,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/conversation-disentanglement-on-irc","task":"Conversation Disentanglement","dataset_variant":"irc-disentanglement","rows":5,"metrics":["VI","1-1","P","R","F"],"first_row_in_archive_order":{"model":"BERT + BiLSTM","paper":"/paper/pre-trained-and-attention-based-neural","metrics":{"F":"46.8","P":"44.3","R":"49.6","VI":"93.3"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/conversation-disentanglement-on-linux-irc-ch2","task":"Conversation Disentanglement","dataset_variant":"Linux IRC (Ch2 Kummerfeld)","rows":3,"metrics":["1-1","Local","Shen F-1"],"first_row_in_archive_order":{"model":"Linear","paper":"/paper/you-talking-to-me-a-corpus-and-algorithm-for","metrics":{"1-1":"59.7","Local":"80.8","Shen F-1":"63.0"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/conversation-disentanglement-on-linux-irc-ch2-1","task":"Conversation Disentanglement","dataset_variant":"Linux IRC (Ch2 Elsner)","rows":3,"metrics":["1-1","Local","Shen F-1"],"first_row_in_archive_order":{"model":"Linear","paper":"/paper/you-talking-to-me-a-corpus-and-algorithm-for","metrics":{"1-1":"53.1","Local":"81.9","Shen F-1":"55.1"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/pre-trained-and-attention-based-neural","title":"Pre-Trained and Attention-Based Neural Networks for Building Noetic Task-Oriented Dialogue Systems","date":"2020-04-04","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/analyzing-assumptions-in-conversation","title":"A Large-Scale Corpus for Conversation Disentanglement","date":"2018-10-25","rows_on_this_dataset":5,"code_links":3,"syntology":null},{"paper":"/paper/training-end-to-end-dialogue-systems-with-the","title":"Training End-to-End Dialogue Systems with the Ubuntu Dialogue Corpus","date":"2017-01-01","rows_on_this_dataset":2,"code_links":0,"syntology":null},{"paper":"/paper/you-talking-to-me-a-corpus-and-algorithm-for","title":"You Talking to Me? A Corpus and Algorithm for Conversation Disentanglement","date":"2008-06-01","rows_on_this_dataset":3,"code_links":0,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}