{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/chat","entry":"chat","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":11,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":7,"n_samples_fingerprinted":0,"n_places":11,"n_places_pointer_only":5,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":5,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.00256","paper":"/paper/arxiv-2609-00256","title":"NSIDDx: A Design Framework for Neuro-Symbolic, Practitioner-First Differential Diagnosis in Low-Resource Settings","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"joetheguide2/NSIDDX-","path":"llm_client.py","file_url":"https://github.com/joetheguide2/NSIDDX-/blob/HEAD/llm_client.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ffcdcf038ed5db09","mcp_get_code":{"code_sha256":"ffcdcf038ed5db09"}},{"arxiv_id":"2604.21564","paper":"/paper/arxiv-2604-21564","title":"Measuring Opinion Bias and Sycophancy via LLM-based Persuasion","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"maritaca-ai/llm-bias-bench","path":"bias_bench.py","file_url":"https://github.com/maritaca-ai/llm-bias-bench/blob/HEAD/bias_bench.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8e661fcefd3cd952","mcp_get_code":{"code_sha256":"8e661fcefd3cd952"}},{"arxiv_id":"2503.14021","paper":"/paper/mp-gui-modality-perception-with-mllms-for-gui","title":"MP-GUI: Modality Perception with MLLMs for GUI Understanding","date":"2025-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BigTaige/MP-GUI","path":"pipeline/vllm_pipeline_v2.py","file_url":"https://github.com/BigTaige/MP-GUI/blob/HEAD/pipeline/vllm_pipeline_v2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dd863a03e032ee31","mcp_get_code":{"code_sha256":"dd863a03e032ee31"}},{"arxiv_id":"2410.04197","paper":"/paper/cs4-measuring-the-creativity-of-large","title":"CS4: Measuring the Creativity of Large Language Models Automatically by Controlling the Number of Story-Writing Constraints","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"anirudhlakkaraju/cs4_benchmark","path":"evaluation/story_quality_eval.py","file_url":"https://github.com/anirudhlakkaraju/cs4_benchmark/blob/HEAD/evaluation/story_quality_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9b2db3fdfca284ab","mcp_get_code":{"code_sha256":"9b2db3fdfca284ab"}},{"arxiv_id":"2410.00927","paper":"/paper/text-clustering-as-classification-with-llms","title":"Text Clustering as Classification with LLMs","date":"2024-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ecnu-text-computing/text-clustering-via-llm","path":"label_generation.py","file_url":"https://github.com/ecnu-text-computing/text-clustering-via-llm/blob/HEAD/label_generation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c7c5f4efa80e3754","mcp_get_code":{"code_sha256":"c7c5f4efa80e3754"}},{"arxiv_id":"2408.15787","paper":"/paper/interactive-agents-simulating-counselor","title":"Interactive Agents: Simulating Counselor-Client Psychological Counseling via Role-Playing LLM-to-LLM Interactions","date":"2024-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qiuhuachuan/interactive-agents","path":"auto_chat_LLM_as_a_judge.py","file_url":"https://github.com/qiuhuachuan/interactive-agents/blob/HEAD/auto_chat_LLM_as_a_judge.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8652f9f8824b6b40","mcp_get_code":{"code_sha256":"8652f9f8824b6b40"}},{"arxiv_id":"2405.14125","paper":"/paper/ali-agent-assessing-llms-alignment-with-human","title":"ALI-Agent: Assessing LLMs' Alignment with Human Values via Agent-based Evaluation","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sophiezheng998/ali-agent","path":"simulation/utils.py","file_url":"https://github.com/sophiezheng998/ali-agent/blob/HEAD/simulation/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6bc98d55a886b87d","mcp_get_code":{"code_sha256":"6bc98d55a886b87d"}},{"arxiv_id":"2403.02076","paper":"/paper/vtg-gpt-tuning-free-zero-shot-video-temporal-1","title":"VTG-GPT: Tuning-Free Zero-Shot Video Temporal Grounding with GPT","date":"2024-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YoucanBaby/VTG-GPT","path":"Baichuan2/rephrase_query.py","file_url":"https://github.com/YoucanBaby/VTG-GPT/blob/HEAD/Baichuan2/rephrase_query.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2de73df87fce327d","mcp_get_code":{"code_sha256":"2de73df87fce327d"}},{"arxiv_id":"2310.15100","paper":"/paper/llm-in-the-loop-leveraging-large-language","title":"LLM-in-the-loop: Leveraging Large Language Model for Thematic Analysis","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sjdai/llm-thematic-analysis","path":"chat_mods.py","file_url":"https://github.com/sjdai/llm-thematic-analysis/blob/HEAD/chat_mods.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef915f77bf55b822","mcp_get_code":{"code_sha256":"ef915f77bf55b822"}},{"arxiv_id":"2310.01386","paper":"/paper/who-is-chatgpt-benchmarking-llms","title":"Who is ChatGPT? Benchmarking LLMs' Psychological Portrayal Using PsychoBench","date":"2023-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cuhk-arise/psychobench","path":"example_generator.py","file_url":"https://github.com/cuhk-arise/psychobench/blob/HEAD/example_generator.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"83f3e946d3bd96b6","mcp_get_code":{"code_sha256":"83f3e946d3bd96b6"}},{"arxiv_id":"2308.09729","paper":"/paper/mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wyl-willing/MindMap","path":"pre-training/explainpe/keyword.py","file_url":"https://github.com/wyl-willing/MindMap/blob/HEAD/pre-training/explainpe/keyword.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b759881245a90f3e","mcp_get_code":{"code_sha256":"b759881245a90f3e"}}]}