{"url":"/dataset/bbai-dataset","name":"BBAI Dataset","full_name":"Black-box Agent Integration","description_markdown":"This dataset is for evaluating the task of Black-box Multi-agent Integration which focuses on combining the capabilities of multiple black-box conversational agents at scale. It provides data to explore two main frameworks of exploration: question agent pairing and question response pairing.\r\n\r\nOverall this dataset contains 5550 utterances with 19 question-response pairs per question (one from each of the 19 agents), 105,450 in total across 37 domains. The utterances are split into 3700 utterances (100 examples per domain) for the training set and 1850 (50 per domain) for the test set. The train and test sets respectively contain 2399 and 1186 utterances with at least one positive question-response pair. In the remaining examples, none of the agents were able to achieve annotator agreement (>= 3).","description_withheld":null,"homepage":"https://github.com/ChrisIsKing/black-box-multi-agent-integation/tree/main/data","introduced_date":"2022-03-15","introduced_date_note":null,"introduced_by":{"paper":"/paper/one-agent-to-rule-them-all-towards-multi-1","title":"One Agent To Rule Them All: Towards Multi-agent Conversational AI","first_author":"Christopher Clarke","url":null},"license":{"name":"CC BY","url":null},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Conversational Response Selection","url":"/task/conversational-response-selection","datasets_with_task":"/datasets/task/conversational-response-selection"},{"name":"Multi-agent Integration","url":"/task/multi-agent-integration","datasets_with_task":"/datasets/task/multi-agent-integration"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["BBAI Dataset"],"data_loaders":[],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/multi-agent-integration-on-bbai-dataset","task":"Multi-agent Integration","dataset_variant":"BBAI Dataset","rows":1,"metrics":["P@1"],"first_row_in_archive_order":{"model":"MARS Encoder","paper":"/paper/one-agent-to-rule-them-all-towards-multi-1","metrics":{"P@1":"83.55"},"code_links":[{"title":"ChrisIsKing/black-box-multi-agent-integation","url":"https://github.com/ChrisIsKing/black-box-multi-agent-integation"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/one-agent-to-rule-them-all-towards-multi-1","title":"One Agent To Rule Them All: Towards Multi-agent Conversational AI","date":"2022-03-15","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}