{"url":"/task/science-question-answering","name":"Science Question Answering","slug":"science-question-answering","description_markdown":"Image credit: [Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering](https://paperswithcode.com/paper/learn-to-explain-multimodal-reasoning-via)","categories":[{"name":"Miscellaneous","url":"/area/miscellaneous"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Reasoning","url":"/area/reasoning"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":23,"papers_with_code":15,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/science-question-answering-on-scienceqa","slug":"science-question-answering-on-scienceqa","dataset":"ScienceQA","dataset_url":"/dataset/scienceqa","rows_in_archive":10,"metrics":["Avg. Accuracy","Natural Science","Social Science","Language Science","Text Context","Image Context","No Context","Grades 1-6","Grades 7-12"],"first_row_in_archive_order":{"model":"MC-CoT F-Large","paper_title":"Boosting the Power of Small Multimodal Reasoning Models to Match Larger Models with Self-Consistency Training","paper_url":"/paper/boosting-the-power-of-small-multimodal","paper_date":"2023-11-23","arxiv_id":"2311.14109","code_links":[{"title":"chengtan9907/mc-cot","url":"https://github.com/chengtan9907/mc-cot"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}}}],"datasets":[{"url":"/dataset/scienceqa","name":"ScienceQA","full_name":"Science Question Answering","num_papers_in_archive":339},{"url":"/dataset/frenchmedmcqa","name":"FrenchMedMCQA","full_name":"FrenchMedMCQA: A French Multiple-Choice Question Answering Dataset for Medical domain","num_papers_in_archive":6},{"url":"/dataset/semopenalex","name":"SemOpenAlex","full_name":"","num_papers_in_archive":6},{"url":"/dataset/biofuelqr","name":"BioFuelQR","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/question-answering","name":"Question Answering"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":15,"of":15,"tagged_in_all":23,"items":[{"url":"/paper/chat-univi-unified-visual-representation","title":"Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding","date":"2023-11-14","arxiv_id":"2311.08046","repositories_listed":4,"syntology":null},{"url":"/paper/multimodal-chain-of-thought-reasoning-in","title":"Multimodal Chain-of-Thought Reasoning in Language Models","date":"2023-02-02","arxiv_id":"2302.00923","repositories_listed":3,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":1}},{"url":"/paper/towards-causalgpt-a-multi-agent-approach-for","title":"Towards CausalGPT: A Multi-Agent Approach for Faithful Knowledge Reasoning via Promoting Causal Consistency in LLMs","date":"2023-08-23","arxiv_id":"2308.11914","repositories_listed":2,"syntology":null},{"url":"/paper/sciqag-a-framework-for-auto-generated","title":"SciQAG: A Framework for Auto-Generated Science Question Answering Dataset with Fine-grained Evaluation","date":"2024-05-16","arxiv_id":"2405.09939","repositories_listed":1,"syntology":null},{"url":"/paper/video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","arxiv_id":"2402.03161","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":5}},{"url":"/paper/honeybee-locality-enhanced-projector-for","title":"Honeybee: Locality-enhanced Projector for Multimodal LLM","date":"2023-12-11","arxiv_id":"2312.06742","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/boosting-the-power-of-small-multimodal","title":"Boosting the Power of Small Multimodal Reasoning Models to Match Larger Models with Self-Consistency Training","date":"2023-11-23","arxiv_id":"2311.14109","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/a-survey-on-interpretable-cross-modal","title":"A Survey on Interpretable Cross-modal Reasoning","date":"2023-09-05","arxiv_id":"2309.01955","repositories_listed":1,"syntology":null},{"url":"/paper/cheap-and-quick-efficient-vision-language","title":"Cheap and Quick: Efficient Vision-Language Instruction Tuning for Large Language Models","date":"2023-05-24","arxiv_id":"2305.15023","repositories_listed":1,"syntology":null},{"url":"/paper/t-sciq-teaching-multimodal-chain-of-thought","title":"T-SciQ: Teaching Multimodal Chain-of-Thought Reasoning via Mixed Large Language Model Signals for Science Question Answering","date":"2023-05-05","arxiv_id":"2305.03453","repositories_listed":1,"syntology":null},{"url":"/paper/two-is-better-than-many-binary-classification","title":"Two is Better than Many? Binary Classification as an Effective Approach to Multi-Choice Question Answering","date":"2022-10-29","arxiv_id":"2210.16495","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/learn-to-explain-multimodal-reasoning-via","title":"Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering","date":"2022-09-20","arxiv_id":"2209.09513","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/exploiting-reasoning-chains-for-multi-hop","title":"Exploiting Reasoning Chains for Multi-hop Science Question Answering","date":"2021-09-07","arxiv_id":"2109.02905","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-semantic-graph-construction-and","title":"Dynamic Semantic Graph Construction and Reasoning for Explainable Multi-hop Science Question Answering","date":"2021-05-25","arxiv_id":"2105.11776","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/unification-based-reconstruction-of","title":"Unification-based Reconstruction of Multi-hop Explanations for Science Questions","date":"2020-03-31","arxiv_id":"2004.00061","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":1,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}