{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/arxiv-2602-01885","title":"ES-MemEval: Benchmarking Conversational Agents on Personalized Long-Term Emotional Support","arxiv_id":"2602.01885","date":"2026-02-02","proceeding":null,"authors":["Tiantian Chen","Jiaqi Lu","Ying Shen","Lin Zhang"],"abstract":"Large Language Models (LLMs) have shown strong potential as conversational agents. Yet, their effectiveness remains limited by deficiencies in robust long-term memory, particularly in complex, long-term web-based services such as online emotional support. However, existing long-term dialogue benchmarks primarily focus on static and explicit fact retrieval, failing to evaluate agents in critical scenarios where user information is dispersed, implicit, and continuously evolving. To address this gap, we introduce ES-MemEval, a comprehensive benchmark that systematically evaluates five core memory capabilities: information extraction, temporal reasoning, conflict detection, abstention, and user modeling, in long-term emotional support settings, covering question answering, summarization, and dialogue generation tasks. To support the benchmark, we also propose EvoEmo, a multi-session dataset for personalized long-term emotional support that captures fragmented, implicit user disclosures and evolving user states. Extensive experiments on open-source long-context, commercial, and retrieval-augmented (RAG) LLMs show that explicit long-term memory is essential for reducing hallucinations and enabling effective personalization. At the same time, RAG improves factual consistency but struggles with temporal dynamics and evolving user states. These findings highlight both the potential and limitations of current paradigms and motivate more robust integration of memory and retrieval for long-term personalized dialogue systems.","url_abs":"https://arxiv.org/abs/2602.01885","url_pdf":"https://arxiv.org/pdf/2602.01885","source":{"archive":null,"snapshot":"2025-07-28","note":"not in the Papers with Code archive (frozen at the snapshot)","row_kind":"graph","title_abstract_authors_date":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)"},"code_links":[],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2602.01885","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2602.01885"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"mentioned_in_github":null,"is_official":null,"provenance":"deterministic:regex_extraction","mentioned_in_paper":null,"url":"https://github.com/slptongji/ES-MemEval","reach":null}],"summary":{"ran":2,"unverified":2},"by_repo_kind":{"found_in_text":{"samples":4,"ran":2,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"ae5829cfa1d8a22a","entry":"RoundWiseMemoryInplaceStrategy","repo":"slptongji/ES-MemEval","repo_kind":"found_in_text","path":"src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","file_url":"https://github.com/slptongji/ES-MemEval/blob/HEAD/src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"ae5829cfa1d8a22a"}},{"code_sha256_prefix":"a70def130b3588a6","entry":"SessionInformation","repo":"slptongji/ES-MemEval","repo_kind":"found_in_text","path":"src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","file_url":"https://github.com/slptongji/ES-MemEval/blob/HEAD/src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"a70def130b3588a6"}},{"code_sha256_prefix":"2fafba04b9452d4f","entry":"DocumentStore","repo":"slptongji/ES-MemEval","repo_kind":"found_in_text","path":"src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","file_url":"https://github.com/slptongji/ES-MemEval/blob/HEAD/src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"2fafba04b9452d4f"}},{"code_sha256_prefix":"7e160fd7825f2acd","entry":"PromptStrategy","repo":"slptongji/ES-MemEval","repo_kind":"found_in_text","path":"src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","file_url":"https://github.com/slptongji/ES-MemEval/blob/HEAD/src/lib/shared/prompt_strategies/round_wise_memory_inplace_strategy.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"7e160fd7825f2acd"}}]},"arxiv_metadata":{"licence":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license)","fields":["title","abstract","authors","date"],"primary_category":"cs.CL","source":"arxiv_2026.jsonl"},"syntology_extracted_results":null}