{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/arxiv-2508-06361","title":"Beyond Prompt-Induced Lies: Investigating LLM Deception on Benign Prompts","arxiv_id":"2508.06361","date":"2025-08-08","proceeding":"ICLR","authors":["Zhaomin Wu","Mingzhe Du","See-Kiong Ng","Bingsheng He"],"abstract":"Large Language Models (LLMs) are widely deployed in reasoning, planning, and decision-making tasks, making their trustworthiness critical. A significant and underexplored risk is intentional deception, where an LLM deliberately fabricates or conceals information to serve a hidden objective. Existing studies typically induce deception by explicitly setting a hidden objective through prompting or fine-tuning, which may not reflect real-world human-LLM interactions. Moving beyond such human-induced deception, we investigate LLMs' self-initiated deception on benign prompts. To address the absence of ground truth, we propose a framework based on Contact Searching Questions (CSQ). This framework introduces two statistical metrics derived from psychological principles to quantify the likelihood of deception. The first, the Deceptive Intention Score, measures the model's bias toward a hidden objective. The second, the Deceptive Behavior Score, measures the inconsistency between the LLM's internal belief and its expressed output. Evaluating 16 leading LLMs, we find that both metrics rise in parallel and escalate with task difficulty for most models. Moreover, increasing model capacity does not always reduce deception, posing a significant challenge for future LLM development.","url_abs":"https://arxiv.org/abs/2508.06361","url_pdf":"https://arxiv.org/pdf/2508.06361","source":{"archive":null,"snapshot":"2025-07-28","note":"not in the Papers with Code archive (frozen at the snapshot)","row_kind":"graph","title_abstract_authors_date":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)"},"code_links":[],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2508.06361","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2508.06361"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"mentioned_in_github":null,"is_official":null,"provenance":"deterministic:regex_extraction","mentioned_in_paper":null,"url":"https://github.com/Xtra-Computing/LLM-Deception","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"unverified":15},"by_repo_kind":{"found_in_text":{"samples":15,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"0f2f1567bae36779","entry":"auto_output_path","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/problem/gen_broken_linkedlist_problems.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/problem/gen_broken_linkedlist_problems.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"0f2f1567bae36779"}},{"code_sha256_prefix":"4ca7eb0f5169dd97","entry":"auto_output_path","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/problem/gen_linkedlist_problems.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/problem/gen_linkedlist_problems.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"4ca7eb0f5169dd97"}},{"code_sha256_prefix":"2a77d7d73c34539e","entry":"calculate_cost","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/ask_openai.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/ask_openai.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"2a77d7d73c34539e"}},{"code_sha256_prefix":"524fed6fd2309786","entry":"check_answer_correctness","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/batch_ask_google.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/batch_ask_google.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"524fed6fd2309786"}},{"code_sha256_prefix":"2cfe7b80a254a5d8","entry":"check_answer_correctness","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/batch_ask_openai.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/batch_ask_openai.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"2cfe7b80a254a5d8"}},{"code_sha256_prefix":"c00c7f258a2353c4","entry":"check_answer_file_exists","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/local_inference.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/local_inference.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"c00c7f258a2353c4"}},{"code_sha256_prefix":"dbca1f5a02eb8f07","entry":"check_answer_file_exists","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/vllm_generate_responses.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/vllm_generate_responses.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"dbca1f5a02eb8f07"}},{"code_sha256_prefix":"f325f37cde5f70c7","entry":"estimate_tokens","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/ask_openai.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/ask_openai.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f325f37cde5f70c7"}},{"code_sha256_prefix":"d9a732c50ecfadc2","entry":"extract_embeddings_from_results","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/extract_embeddings_from_csv.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/extract_embeddings_from_csv.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"d9a732c50ecfadc2"}},{"code_sha256_prefix":"26f108de561d83f3","entry":"generate_unique_names","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/problem/gen_names.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/problem/gen_names.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"26f108de561d83f3"}},{"code_sha256_prefix":"9f5cd3898037646d","entry":"load_csv_results","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/extract_embeddings_from_csv.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/extract_embeddings_from_csv.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"9f5cd3898037646d"}},{"code_sha256_prefix":"ba47bd9499eeb0c2","entry":"load_problems","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/local_inference.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/local_inference.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ba47bd9499eeb0c2"}},{"code_sha256_prefix":"4132c8a3a8c91eeb","entry":"parse_csv_filename_params","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/extract_embeddings_from_csv.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/extract_embeddings_from_csv.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"4132c8a3a8c91eeb"}},{"code_sha256_prefix":"ee25c3d8a7979f85","entry":"parse_filename_params","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/ask_openai.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/ask_openai.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"ee25c3d8a7979f85"}},{"code_sha256_prefix":"f35c3addc72bbf30","entry":"parse_filename_params","repo":"Xtra-Computing/LLM-Deception","repo_kind":"found_in_text","path":"src/llm/local_inference.py","file_url":"https://github.com/Xtra-Computing/LLM-Deception/blob/HEAD/src/llm/local_inference.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"f35c3addc72bbf30"}}]},"arxiv_metadata":{"licence":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)","fields":["title","abstract","authors","date"],"primary_category":"cs.LG","source":"arxiv_api"},"syntology_extracted_results":null}