{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/multi-expert-prompting-improves-reliability","title":"Multi-expert Prompting Improves Reliability, Safety, and Usefulness of Large Language Models","arxiv_id":"2411.00492","date":"2024-11-01","proceeding":null,"authors":["Do Xuan Long","Duong Ngoc Yen","Anh Tuan Luu","Kenji Kawaguchi","Min-Yen Kan","Nancy F. Chen"],"abstract":"We present Multi-expert Prompting, a novel enhancement of ExpertPrompting (Xu et al., 2023), designed to improve the large language model (LLM) generation. Specifically, it guides an LLM to fulfill an input instruction by simulating multiple experts, aggregating their responses, and selecting the best among individual and aggregated responses. This process is performed in a single chain of thoughts through our seven carefully designed subtasks derived from the Nominal Group Technique (Ven and Delbecq, 1974), a well-established decision-making framework. Our evaluations demonstrate that Multi-expert Prompting significantly outperforms ExpertPrompting and comparable baselines in enhancing the truthfulness, factuality, informativeness, and usefulness of responses while reducing toxicity and hurtfulness. It further achieves state-of-the-art truthfulness by outperforming the best baseline by 8.69% with ChatGPT. Multi-expert Prompting is efficient, explainable, and highly adaptable to diverse scenarios, eliminating the need for manual prompt construction.","url_abs":"https://arxiv.org/abs/2411.00492v1","url_pdf":"https://arxiv.org/pdf/2411.00492v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"multi-expert-prompting-improves-reliability","repo_url":"https://github.com/dxlong2000/multi-expert-prompting","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[{"task_slug":"decision-making","task_name":"Decision Making"},{"task_slug":"informativeness","task_name":"Informativeness"},{"task_slug":"language-modeling","task_name":"Language Modeling"},{"task_slug":"language-modelling","task_name":"Language Modelling"},{"task_slug":"large-language-model","task_name":"Large Language Model"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2411.00492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00492"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/dxlong2000/multi-expert-prompting","reach":null}],"summary":{"ran_draft_wrong":5,"unverified":1},"by_repo_kind":{"official":{"samples":6,"ran":5,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":6,"samples":[{"code_sha256_prefix":"f7550082d3e2c065","entry":"generate_answer_choice","repo":"dxlong2000/multi-expert-prompting","repo_kind":"official","path":"src/mep.py","file_url":"https://github.com/dxlong2000/multi-expert-prompting/blob/HEAD/src/mep.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f7550082d3e2c065"}},{"code_sha256_prefix":"702f21f4ae2aed88","entry":"generate_answer_format","repo":"dxlong2000/multi-expert-prompting","repo_kind":"official","path":"src/mep.py","file_url":"https://github.com/dxlong2000/multi-expert-prompting/blob/HEAD/src/mep.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"702f21f4ae2aed88"}},{"code_sha256_prefix":"61f81ac3bf834d7d","entry":"generate_role_format","repo":"dxlong2000/multi-expert-prompting","repo_kind":"official","path":"src/mep.py","file_url":"https://github.com/dxlong2000/multi-expert-prompting/blob/HEAD/src/mep.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"61f81ac3bf834d7d"}},{"code_sha256_prefix":"42916ee04990b77a","entry":"get_llm_answer_with_retry","repo":"dxlong2000/multi-expert-prompting","repo_kind":"official","path":"src/mep.py","file_url":"https://github.com/dxlong2000/multi-expert-prompting/blob/HEAD/src/mep.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"42916ee04990b77a"}},{"code_sha256_prefix":"81b5a86879eee80e","entry":"get_model_answer","repo":"dxlong2000/multi-expert-prompting","repo_kind":"official","path":"src/mep.py","file_url":"https://github.com/dxlong2000/multi-expert-prompting/blob/HEAD/src/mep.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"81b5a86879eee80e"}},{"code_sha256_prefix":"41256578a2413131","entry":"MEP","repo":"dxlong2000/multi-expert-prompting","repo_kind":"official","path":"src/mep.py","file_url":"https://github.com/dxlong2000/multi-expert-prompting/blob/HEAD/src/mep.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"41256578a2413131"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}