{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/tellmewhy-a-dataset-for-answering-why","title":"TellMeWhy: A Dataset for Answering Why-Questions in Narratives","arxiv_id":"2106.06132","date":"2021-06-11","proceeding":"Findings (ACL) 2021 8","authors":["Yash Kumar Lal","Nathanael Chambers","Raymond Mooney","Niranjan Balasubramanian"],"abstract":"Answering questions about why characters perform certain actions is central to understanding and reasoning about narratives. Despite recent progress in QA, it is not clear if existing models have the ability to answer \"why\" questions that may require commonsense knowledge external to the input narrative. In this work, we introduce TellMeWhy, a new crowd-sourced dataset that consists of more than 30k questions and free-form answers concerning why characters in short narratives perform the actions described. For a third of this dataset, the answers are not present within the narrative. Given the limitations of automated evaluation for this task, we also present a systematized human evaluation interface for this dataset. Our evaluation of state-of-the-art models show that they are far below human performance on answering such questions. They are especially worse on questions whose answers are external to the narrative, thus providing a challenge for future QA and narrative understanding research.","url_abs":"https://arxiv.org/abs/2106.06132v3","url_pdf":"https://arxiv.org/pdf/2106.06132v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"tellmewhy-a-dataset-for-answering-why","repo_url":"https://github.com/StonyBrookNLP/tellmewhy","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":null}],"tasks":[],"methods":[],"datasets_introduced":[{"slug":"tellmewhy","name":"TellMeWhy","full_name":""}],"methods_introduced":[],"results":[],"syntology":{"syntology_url":"https://syntology.ai/paper/2106.06132","atlas_url":"https://app.syntology.ai/?focus=2106.06132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06132"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/StonyBrookNLP/tellmewhy","reach":null}],"summary":{"ran_draft_wrong":3,"ran_honours":1,"ran_violates":1,"unverified":1},"by_repo_kind":{"listed":{"samples":6,"ran":5,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":6,"samples":[{"code_sha256_prefix":"5be895a93593b73e","entry":"create_inputs_for_rouge","repo":"StonyBrookNLP/tellmewhy","repo_kind":"listed","path":"src/automatic_evaluation.py","file_url":"https://github.com/StonyBrookNLP/tellmewhy/blob/HEAD/src/automatic_evaluation.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"5be895a93593b73e"}},{"code_sha256_prefix":"0e24b5aef615871d","entry":"create_multi_reference_dictionary_for_gold_sentences","repo":"StonyBrookNLP/tellmewhy","repo_kind":"listed","path":"src/automatic_evaluation.py","file_url":"https://github.com/StonyBrookNLP/tellmewhy/blob/HEAD/src/automatic_evaluation.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0e24b5aef615871d"}},{"code_sha256_prefix":"16b36baf8c9bc5c8","entry":"extract_true_annotations","repo":"StonyBrookNLP/tellmewhy","repo_kind":"listed","path":"src/analyse_human_evaluation_output.py","file_url":"https://github.com/StonyBrookNLP/tellmewhy/blob/HEAD/src/analyse_human_evaluation_output.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"16b36baf8c9bc5c8"}},{"code_sha256_prefix":"cb3b5d6e87cb16b5","entry":"fleiss_kappa","repo":"StonyBrookNLP/tellmewhy","repo_kind":"listed","path":"src/analyse_human_evaluation_output.py","file_url":"https://github.com/StonyBrookNLP/tellmewhy/blob/HEAD/src/analyse_human_evaluation_output.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"cb3b5d6e87cb16b5"}},{"code_sha256_prefix":"f3025cd8b5ffa609","entry":"sentence_level_multi_bertscore","repo":"StonyBrookNLP/tellmewhy","repo_kind":"listed","path":"src/automatic_evaluation.py","file_url":"https://github.com/StonyBrookNLP/tellmewhy/blob/HEAD/src/automatic_evaluation.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f3025cd8b5ffa609"}},{"code_sha256_prefix":"0eec09974e44d9dd","entry":"weighted_fleiss_kappa","repo":"StonyBrookNLP/tellmewhy","repo_kind":"listed","path":"src/analyse_human_evaluation_output.py","file_url":"https://github.com/StonyBrookNLP/tellmewhy/blob/HEAD/src/analyse_human_evaluation_output.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"0eec09974e44d9dd"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}