{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/virtual-personas-for-language-models-via-an","title":"Virtual Personas for Language Models via an Anthology of Backstories","arxiv_id":"2407.06576","date":"2024-07-09","proceeding":null,"authors":["Suhong Moon","Marwa Abdulhai","Minwoo Kang","Joseph Suh","Widyadewi Soedarmadji","Eran Kohen Behar","David M. Chan"],"abstract":"Large language models (LLMs) are trained from vast repositories of text authored by millions of distinct authors, reflecting an enormous diversity of human traits. While these models bear the potential to be used as approximations of human subjects in behavioral studies, prior efforts have been limited in steering model responses to match individual human users. In this work, we introduce \"Anthology\", a method for conditioning LLMs to particular virtual personas by harnessing open-ended life narratives, which we refer to as \"backstories.\" We show that our methodology enhances the consistency and reliability of experimental outcomes while ensuring better representation of diverse sub-populations. Across three nationally representative human surveys conducted as part of Pew Research Center's American Trends Panel (ATP), we demonstrate that Anthology achieves up to 18% improvement in matching the response distributions of human respondents and 27% improvement in consistency metrics. Our code and generated backstories are available at https://github.com/CannyLab/anthology.","url_abs":"https://arxiv.org/abs/2407.06576v3","url_pdf":"https://arxiv.org/pdf/2407.06576v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"virtual-personas-for-language-models-via-an","repo_url":"https://github.com/cannylab/anthology","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"BSD-3-Clause"}}],"tasks":[{"task_slug":"diversity","task_name":"Diversity"}],"methods":[{"method_slug":null,"method_name":"American"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2407.06576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06576"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/cannylab/anthology","reach":{"status":"ok","spdx":"BSD-3-Clause"}},{"provenance":"deterministic:regex_extraction","url":"https://github.com/CannyLab/anthology","reach":{"status":"ok","spdx":"BSD-3-Clause"}}],"summary":{"ran":9,"ran_draft_wrong":1,"unverified":5},"by_repo_kind":{"official":{"samples":15,"ran":10,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"2216c17ba1c6e4b5","entry":"generate_mcq","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/lm_inference/question_utils.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/lm_inference/question_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"2216c17ba1c6e4b5"}},{"code_sha256_prefix":"9832a0e52b22286b","entry":"label_to_index_mapper","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/response_parse_utils.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/response_parse_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"9832a0e52b22286b"}},{"code_sha256_prefix":"1169149be5b9f40b","entry":"load_backstory","repo":"CannyLab/anthology","repo_kind":"official","path":"anthology/backstory/backstory.py","file_url":"https://github.com/CannyLab/anthology/blob/HEAD/anthology/backstory/backstory.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"1169149be5b9f40b"}},{"code_sha256_prefix":"6a7644fed88642f7","entry":"load_data","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/analysis/atp.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/analysis/atp.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"6a7644fed88642f7"}},{"code_sha256_prefix":"515607e16f43d17c","entry":"nonzero_index_finder","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/response_parse_utils.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/response_parse_utils.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"515607e16f43d17c"}},{"code_sha256_prefix":"09aa43be3289a19d","entry":"prepare_llm","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/lm_inference/llm.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/lm_inference/llm.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"09aa43be3289a19d"}},{"code_sha256_prefix":"393f36298ef6a632","entry":"regex_leaning_letter_classifier","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/political_affiliation.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/political_affiliation.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"393f36298ef6a632"}},{"code_sha256_prefix":"a536ee7b78d018e3","entry":"regex_strength_letter_classifier","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/political_affiliation.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/political_affiliation.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"a536ee7b78d018e3"}},{"code_sha256_prefix":"0d42cc0269a379e6","entry":"regex_strength_response_classifier","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/political_affiliation.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/political_affiliation.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"0d42cc0269a379e6"}},{"code_sha256_prefix":"052a7adf33f0375f","entry":"retry_with_exponential_backoff","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/lm_inference/backoff.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/lm_inference/backoff.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"052a7adf33f0375f"}},{"code_sha256_prefix":"c3a7b5e25aa5107e","entry":"get_tokenizer","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/lm_inference/llm.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/lm_inference/llm.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"c3a7b5e25aa5107e"}},{"code_sha256_prefix":"305afac82adc9356","entry":"number_to_index_mapper","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/response_parse_utils.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/response_parse_utils.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"305afac82adc9356"}},{"code_sha256_prefix":"2282f20fccb0175c","entry":"numerical_response_parser","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/response_parser.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/response_parser.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"2282f20fccb0175c"}},{"code_sha256_prefix":"41533974dab8e04c","entry":"preprocess_numerical_response","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/response_parser.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/response_parser.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"41533974dab8e04c"}},{"code_sha256_prefix":"320efa985e8b290b","entry":"string_response_parser","repo":"cannylab/anthology","repo_kind":"official","path":"anthology/demographic_survey/response_parser.py","file_url":"https://github.com/cannylab/anthology/blob/HEAD/anthology/demographic_survey/response_parser.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"mcp_get_code":{"code_sha256":"320efa985e8b290b"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}