{"url":"/dataset/business-scene-dialogue","name":"Business Scene Dialogue","full_name":null,"description_markdown":"The Japanese-English business conversation corpus, namely **Business Scene Dialogue** corpus, was constructed in 3 steps:\r\n\r\n1. selecting business scenes,\r\n2. writing monolingual conversation scenarios according to the selected scenes, and\r\n3. translating the scenarios into the other language.\r\n\r\nHalf of the monolingual scenarios were written in Japanese and the other half were written in English. The whole construction process was supervised by a person who satisfies the following conditions to guarantee the conversations to be natural:\r\n\r\n- has the experience of being engaged in language learning programs, especially for business conversations\r\n- is able to smoothly communicate with others in various business scenes both in Japanese and English\r\n- has the experience of being involved in business\r\n\r\nThe BSD corpus is split into balanced training, development and evaluation sets. The documents in these sets are balanced in terms of scenes and original languages. In this repository we publicly share the full development and evaluation sets and a part of the training data set.\r\n\r\nSource: [BSD](https://github.com/tsuruoka-lab/BSD)","description_withheld":null,"homepage":"https://github.com/tsuruoka-lab/BSD","introduced_date":"2020-08-05","introduced_date_note":null,"introduced_by":{"paper":"/paper/designing-the-business-conversation-corpus-1","title":"Designing the Business Conversation Corpus","first_author":"Matīss Rikters","url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"}],"languages":[{"name":"English","url":"/datasets/language/english"},{"name":"Japanese","url":"/datasets/language/japanese"}],"variants":["Business Scene Dialogue JA-EN","Business Scene Dialogue EN-JA","Business Scene Dialogue"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/ryo0634/bsd_ja_en","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/bsd_ja_en","frameworks":["tf","pytorch","jax"]},{"repo":"https://github.com/tsuruoka-lab/BSD","url":"https://github.com/tsuruoka-lab/BSD","frameworks":[]}],"num_papers_in_archive":9,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/machine-translation-on-business-scene","task":"Machine Translation","dataset_variant":"Business Scene Dialogue JA-EN","rows":1,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"Transformer-base","paper":"/paper/designing-the-business-conversation-corpus-1","metrics":{"BLEU":"12.88"},"code_links":[{"title":"tsuruoka-lab/BSD","url":"https://github.com/tsuruoka-lab/BSD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/machine-translation-on-business-scene-1","task":"Machine Translation","dataset_variant":"Business Scene Dialogue EN-JA","rows":1,"metrics":["BLEU"],"first_row_in_archive_order":{"model":"Transformer-base","paper":"/paper/designing-the-business-conversation-corpus-1","metrics":{"BLEU":"13.53"},"code_links":[{"title":"tsuruoka-lab/BSD","url":"https://github.com/tsuruoka-lab/BSD"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/designing-the-business-conversation-corpus-1","title":"Designing the Business Conversation Corpus","date":"2020-08-05","rows_on_this_dataset":2,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}