{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/25","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":25,"pages_in_order":29,"rows_per_page":100,"rows":[2401,2500],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/24","next":"/method/gpt-4/papers/26","papers":[{"paper":null,"slug":"aligning-large-multimodal-models-with","title":"Aligning Large Multimodal Models with Factually Augmented RLHF","date":"2023-09-25","arxiv_id":"2309.14525","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-cognitive-maps-and-planning-in","title":"Evaluating Cognitive Maps and Planning in Large Language Models with CogEval","date":"2023-09-25","arxiv_id":"2309.15129","n_code_links":0,"syntology":null},{"paper":null,"slug":"guess-sketch-language-model-guided","title":"Guess & Sketch: Language Model Guided Transpilation","date":"2023-09-25","arxiv_id":"2309.14396","n_code_links":0,"syntology":null},{"paper":null,"slug":"physics-of-language-models-part-3-2-knowledge","title":"Physics of Language Models: Part 3.2, Knowledge Manipulation","date":"2023-09-25","arxiv_id":"2309.14402","n_code_links":0,"syntology":null},{"paper":null,"slug":"watch-your-language-large-language-models-and","title":"Watch Your Language: Investigating Content Moderation with Large Language Models","date":"2023-09-25","arxiv_id":"2309.14517","n_code_links":0,"syntology":null},{"paper":null,"slug":"natural-language-based-context-modeling-and","title":"Natural Language based Context Modeling and Reasoning for Ubiquitous Computing with Large Language Models: A Tutorial","date":"2023-09-24","arxiv_id":"2309.15074","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-chat-about-boring-problems-studying-gpt","title":"A Chat About Boring Problems: Studying GPT-based text normalization","date":"2023-09-23","arxiv_id":"2309.13426","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-large-language-models-cognitive","title":"Probing the Moral Development of Large Language Models through Defining Issues Test","date":"2023-09-23","arxiv_id":"2309.13356","n_code_links":0,"syntology":null},{"paper":"/paper/glotscript-a-resource-and-tool-for-low","slug":"glotscript-a-resource-and-tool-for-low","title":"GlotScript: A Resource and Tool for Low Resource Writing System Identification","date":"2023-09-23","arxiv_id":"2309.13320","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["cisnlp/GlotScript"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":"/paper/reconcile-round-table-conference-improves","slug":"reconcile-round-table-conference-improves","title":"ReConcile: Round-Table Conference Improves Reasoning via Consensus among Diverse LLMs","date":"2023-09-22","arxiv_id":"2309.13007","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dinobby/reconcile"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/acegpt-localizing-large-language-models-in","slug":"acegpt-localizing-large-language-models-in","title":"AceGPT, Localizing Large Language Models in Arabic","date":"2023-09-21","arxiv_id":"2309.12053","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-llms-augment-low-resource-reading","title":"Can LLMs Augment Low-Resource Reading Comprehension Datasets? Opportunities and Challenges","date":"2023-09-21","arxiv_id":"2309.12426","n_code_links":0,"syntology":null},{"paper":"/paper/code-soliloquies-for-accurate-calculations-in","slug":"code-soliloquies-for-accurate-calculations-in","title":"Code Soliloquies for Accurate Calculations in Large Language Models","date":"2023-09-21","arxiv_id":"2309.12161","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["luffycodes/tutorbot-spock-phys"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"llmr-real-time-prompting-of-interactive","title":"LLMR: Real-time Prompting of Interactive Worlds using Large Language Models","date":"2023-09-21","arxiv_id":"2309.12276","n_code_links":0,"syntology":null},{"paper":"/paper/lmsys-chat-1m-a-large-scale-real-world-llm","slug":"lmsys-chat-1m-a-large-scale-real-world-llm","title":"LMSYS-Chat-1M: A Large-Scale Real-World LLM Conversation Dataset","date":"2023-09-21","arxiv_id":"2309.11998","n_code_links":5,"syntology":null},{"paper":null,"slug":"michao-huafen-1-0-a-specialized-pre-trained","title":"MiChao-HuaFen 1.0: A Specialized Pre-trained Corpus Dataset for Domain-specific Large Models","date":"2023-09-21","arxiv_id":"2309.13079","n_code_links":0,"syntology":null},{"paper":null,"slug":"multimodal-deep-learning-for-scientific","title":"Multimodal Deep Learning for Scientific Imaging Interpretation","date":"2023-09-21","arxiv_id":"2309.12460","n_code_links":0,"syntology":null},{"paper":"/paper/random-access-infinite-context-length-for","slug":"random-access-infinite-context-length-for","title":"Random-Access Infinite Context Length for Transformers","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"spring-studying-papers-and-reasoning-to-play","title":"SPRING: Studying Papers and Reasoning to play Games","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/the-cambridge-law-corpus-a-corpus-for-legal-1","slug":"the-cambridge-law-corpus-a-corpus-for-legal-1","title":"The Cambridge Law Corpus: A Dataset for Legal AI Research","date":"2023-09-21","arxiv_id":"2309.12269","n_code_links":0,"syntology":null},{"paper":"/paper/the-reversal-curse-llms-trained-on-a-is-b","slug":"the-reversal-curse-llms-trained-on-a-is-b","title":"The Reversal Curse: LLMs trained on \"A is B\" fail to learn \"B is A\"","date":"2023-09-21","arxiv_id":"2309.12288","n_code_links":2,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["lukasberglund/reversal_curse"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"toward-re-identifying-any-animal","title":"Toward Re-Identifying Any Animal","date":"2023-09-21","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-in-mafia-like-game-simulation","title":"Generative AI in Mafia-like Game Simulation","date":"2023-09-20","arxiv_id":"2309.11672","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-gpt4-a-good-trader","title":"Is GPT4 a Good Trader?","date":"2023-09-20","arxiv_id":"2309.10982","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-evaluation-of-gpt-4-on-the-ethics-dataset","title":"An Evaluation of GPT-4 on the ETHICS Dataset","date":"2023-09-19","arxiv_id":"2309.10492","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-self-reinforcement-for-improving","slug":"exploring-self-reinforcement-for-improving","title":"Exploring Iterative Enhancement for Improving Learnersourced Multiple-Choice Question Explanations with Large Language Models","date":"2023-09-19","arxiv_id":"2309.10444","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["strong-ai-lab/explanation-generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-ai-vs-agi-the-cognitive-strengths","title":"Generative AI vs. AGI: The Cognitive Strengths and Weaknesses of Modern LLMs","date":"2023-09-19","arxiv_id":"2309.10371","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-speech-ptm-text-llm-and-emotional","title":"Leveraging Speech PTM, Text LLM, and Emotional TTS for Speech Emotion Recognition","date":"2023-09-19","arxiv_id":"2309.10294","n_code_links":0,"syntology":null},{"paper":"/paper/mint-evaluating-llms-in-multi-turn","slug":"mint-evaluating-llms-in-multi-turn","title":"MINT: Evaluating LLMs in Multi-turn Interaction with Tools and Language Feedback","date":"2023-09-19","arxiv_id":"2309.10691","n_code_links":1,"syntology":null},{"paper":null,"slug":"policygpt-automated-analysis-of-privacy","title":"PolicyGPT: Automated Analysis of Privacy Policies with Large Language Models","date":"2023-09-19","arxiv_id":"2309.10238","n_code_links":0,"syntology":null},{"paper":"/paper/facilitating-nsfw-text-detection-in-open","slug":"facilitating-nsfw-text-detection-in-open","title":"Facilitating NSFW Text Detection in Open-Domain Dialogue Systems via Knowledge Distillation","date":"2023-09-18","arxiv_id":"2309.09749","n_code_links":1,"syntology":null},{"paper":"/paper/embrace-divergence-for-richer-insights-a","slug":"embrace-divergence-for-richer-insights-a","title":"Embrace Divergence for Richer Insights: A Multi-document Summarization Benchmark and a Case Study on Summarizing Diverse Information from News Articles","date":"2023-09-17","arxiv_id":"2309.09369","n_code_links":1,"syntology":null},{"paper":null,"slug":"performance-of-the-pre-trained-large-language","title":"Performance of the Pre-Trained Large Language Model GPT-4 on Automated Short Answer Grading","date":"2023-09-17","arxiv_id":"2309.09338","n_code_links":0,"syntology":null},{"paper":"/paper/examining-the-influence-of-varied-levels-of","slug":"examining-the-influence-of-varied-levels-of","title":"Examining the Influence of Varied Levels of Domain Knowledge Base Inclusion in GPT-based Intelligent Tutors","date":"2023-09-16","arxiv_id":"2309.12367","n_code_links":1,"syntology":null},{"paper":"/paper/struc-bench-are-large-language-models-really","slug":"struc-bench-are-large-language-models-really","title":"Struc-Bench: Are Large Language Models Really Good at Generating Complex Structured Data?","date":"2023-09-16","arxiv_id":"2309.08963","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-lab-next-generation-of-optimal-chemistry","title":"GPT-Lab: Next Generation Of Optimal Chemistry Discovery By GPT Driven Robotic Lab","date":"2023-09-15","arxiv_id":"2309.16721","n_code_links":0,"syntology":null},{"paper":"/paper/iclef-in-context-learning-with-expert","slug":"iclef-in-context-learning-with-expert","title":"ICLEF: In-Context Learning with Expert Feedback for Explainable Style Transfer","date":"2023-09-15","arxiv_id":"2309.08583","n_code_links":1,"syntology":null},{"paper":"/paper/investlm-a-large-language-model-for","slug":"investlm-a-large-language-model-for","title":"InvestLM: A Large Language Model for Investment using Financial Domain Instruction Tuning","date":"2023-09-15","arxiv_id":"2309.13064","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["abacinlp/investlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"openai-cribbed-our-tax-example-but-can-gpt-4","title":"OpenAI Cribbed Our Tax Example, But Can GPT-4 Really Do Tax?","date":"2023-09-15","arxiv_id":"2309.09992","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-large-language-model-based-evaluators-the","title":"Are Large Language Model-based Evaluators the Solution to Scaling Up Multilingual Evaluation?","date":"2023-09-14","arxiv_id":"2309.07462","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai","title":"Generative AI","date":"2023-09-13","arxiv_id":"2309.07930","n_code_links":0,"syntology":null},{"paper":"/paper/in-contextual-bias-suppression-for-large","slug":"in-contextual-bias-suppression-for-large","title":"In-Contextual Gender Bias Suppression for Large Language Models","date":"2023-09-13","arxiv_id":"2309.07251","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-can-infer-psychological","title":"Large Language Models Can Infer Psychological Dispositions of Social Media Users","date":"2023-09-13","arxiv_id":"2309.08631","n_code_links":0,"syntology":null},{"paper":"/paper/rain-your-language-models-can-align","slug":"rain-your-language-models-can-align","title":"RAIN: Your Language Models Can Align Themselves without Finetuning","date":"2023-09-13","arxiv_id":"2309.07124","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["SafeAILab/RAIN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/safetybench-evaluating-the-safety-of-large","slug":"safetybench-evaluating-the-safety-of-large","title":"SafetyBench: Evaluating the Safety of Large Language Models","date":"2023-09-13","arxiv_id":"2309.07045","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thu-coai/safetybench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/2309-06085","slug":"2309-06085","title":"BHASA: A Holistic Southeast Asian Linguistic and Cultural Evaluation Suite for Large Language Models","date":"2023-09-12","arxiv_id":"2309.06085","n_code_links":3,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-and-weak","title":"Leveraging Large Language Models and Weak Supervision for Social Media data annotation: an evaluation using COVID-19 self-reported vaccination tweets","date":"2023-09-12","arxiv_id":"2309.06503","n_code_links":0,"syntology":null},{"paper":null,"slug":"strategic-behavior-of-large-language-models","title":"Strategic Behavior of Large Language Models: Game Structure vs. Contextual Framing","date":"2023-09-12","arxiv_id":"2309.05898","n_code_links":0,"syntology":null},{"paper":"/paper/the-moral-machine-experiment-on-large","slug":"the-moral-machine-experiment-on-large","title":"The Moral Machine Experiment on Large Language Models","date":"2023-09-12","arxiv_id":"2309.05958","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-empirical-study-of-netops-capability-of","title":"An Empirical Study of NetOps Capability of Pre-Trained Large Language Models","date":"2023-09-11","arxiv_id":"2309.05557","n_code_links":0,"syntology":null},{"paper":null,"slug":"black-box-analysis-gpts-across-time-in-legal","title":"Black-Box Analysis: GPTs Across Time in Legal Textual Entailment Task","date":"2023-09-11","arxiv_id":"2309.05501","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-model-for-science-a-study-on-p","slug":"large-language-model-for-science-a-study-on-p","title":"Large Language Model for Science: A Study on P vs. NP","date":"2023-09-11","arxiv_id":"2309.05689","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-finetuning-large-language-models","title":"Efficient Finetuning Large Language Models For Vietnamese Chatbot","date":"2023-09-09","arxiv_id":"2309.04646","n_code_links":0,"syntology":null},{"paper":"/paper/fimo-a-challenge-formal-dataset-for-automated","slug":"fimo-a-challenge-formal-dataset-for-automated","title":"FIMO: A Challenge Formal Dataset for Automated Theorem Proving","date":"2023-09-08","arxiv_id":"2309.04295","n_code_links":1,"syntology":null},{"paper":null,"slug":"from-sparse-to-dense-gpt-4-summarization-with","title":"From Sparse to Dense: GPT-4 Summarization with Chain of Density Prompting","date":"2023-09-08","arxiv_id":"2309.04269","n_code_links":0,"syntology":null},{"paper":"/paper/nestle-a-no-code-tool-for-statistical","slug":"nestle-a-no-code-tool-for-statistical","title":"NESTLE: a No-Code Tool for Statistical Analysis of Legal Corpus","date":"2023-09-08","arxiv_id":"2309.04146","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-pipeline-based-conversational","title":"Enhancing Pipeline-Based Conversational Agents with Large Language Models","date":"2023-09-07","arxiv_id":"2309.03748","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-the-efficacy-of-supervised","slug":"evaluating-the-efficacy-of-supervised","title":"Supervised Learning and Large Language Model Benchmarks on Mental Health Datasets: Cognitive Distortions and Suicidal Risks in Chinese Social Media","date":"2023-09-07","arxiv_id":"2309.03564","n_code_links":2,"syntology":null},{"paper":"/paper/evaluation-of-large-language-models-for","slug":"evaluation-of-large-language-models-for","title":"Evaluation of large language models for discovery of gene set function","date":"2023-09-07","arxiv_id":"2309.04019","n_code_links":1,"syntology":null},{"paper":"/paper/zero-shot-audio-captioning-via-audibility","slug":"zero-shot-audio-captioning-via-audibility","title":"Zero-Shot Audio Captioning via Audibility Guidance","date":"2023-09-07","arxiv_id":"2309.03884","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-can-solve-mathematical-problems-without-a","slug":"gpt-can-solve-mathematical-problems-without-a","title":"GPT Can Solve Mathematical Problems Without a Calculator","date":"2023-09-06","arxiv_id":"2309.03241","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thudm/mathglm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"knowledge-solver-teaching-llms-to-search-for","title":"Knowledge Solver: Teaching LLMs to Search for Domain Knowledge from Knowledge Graphs","date":"2023-09-06","arxiv_id":"2309.03118","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-automated-open","slug":"large-language-models-for-automated-open","title":"Large Language Models for Automated Open-domain Scientific Hypotheses Discovery","date":"2023-09-06","arxiv_id":"2309.02726","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":6,"n_instrument":0,"unverified":5,"pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["zongliny/moose"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/codeapex-a-bilingual-programming-evaluation","slug":"codeapex-a-bilingual-programming-evaluation","title":"CodeApex: A Bilingual Programming Evaluation Benchmark for Large Language Models","date":"2023-09-05","arxiv_id":"2309.01940","n_code_links":1,"syntology":null},{"paper":"/paper/data-juicer-a-one-stop-data-processing-system","slug":"data-juicer-a-one-stop-data-processing-system","title":"Data-Juicer: A One-Stop Data Processing System for Large Language Models","date":"2023-09-05","arxiv_id":"2309.02033","n_code_links":2,"syntology":null},{"paper":null,"slug":"bias-assessment-and-mitigation-in-llm-based","title":"Bias Testing and Mitigation in LLM-based Code Generation","date":"2023-09-03","arxiv_id":"2309.14345","n_code_links":0,"syntology":null},{"paper":"/paper/value-kaleidoscope-engaging-ai-with","slug":"value-kaleidoscope-engaging-ai-with","title":"Value Kaleidoscope: Engaging AI with Pluralistic Human Values, Rights, and Duties","date":"2023-09-02","arxiv_id":"2309.00779","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-for-semantic-monitoring","title":"Large Language Models for Semantic Monitoring of Corporate Disclosures: A Case Study on Korea's Top 50 KOSPI Companies","date":"2023-09-01","arxiv_id":"2309.00208","n_code_links":0,"syntology":null},{"paper":"/paper/publicly-shareable-clinical-large-language","slug":"publicly-shareable-clinical-large-language","title":"Publicly Shareable Clinical Large Language Model Built on Synthetic Clinical Notes","date":"2023-09-01","arxiv_id":"2309.00237","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["starmpcc/asclepius"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/biocoder-a-benchmark-for-bioinformatics-code","slug":"biocoder-a-benchmark-for-bioinformatics-code","title":"BioCoder: A Benchmark for Bioinformatics Code Generation with Large Language Models","date":"2023-08-31","arxiv_id":"2308.16458","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gersteinlab/biocoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-subtask-performance-of-multi-modal","title":"Enhancing Subtask Performance of Multi-modal Large Language Model","date":"2023-08-31","arxiv_id":"2308.16474","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-has-become-financially-literate-insights","title":"GPT has become financially literate: Insights from financial literacy tests of GPT and a preliminary test of how people use it as a source of advice","date":"2023-08-31","arxiv_id":"2309.00649","n_code_links":0,"syntology":null},{"paper":null,"slug":"linking-microblogging-sentiments-to-stock","title":"Linking microblogging sentiments to stock price movement: An application of GPT-4","date":"2023-08-31","arxiv_id":"2308.16771","n_code_links":0,"syntology":null},{"paper":"/paper/touchstone-evaluating-vision-language-models","slug":"touchstone-evaluating-vision-language-models","title":"TouchStone: Evaluating Vision-Language Models by Language Models","date":"2023-08-31","arxiv_id":"2308.16890","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-as-data-preprocessors","title":"Large Language Models as Data Preprocessors","date":"2023-08-30","arxiv_id":"2308.16361","n_code_links":0,"syntology":null},{"paper":null,"slug":"recursively-summarizing-enables-long-term","title":"Recursively Summarizing Enables Long-Term Dialogue Memory in Large Language Models","date":"2023-08-29","arxiv_id":"2308.15022","n_code_links":0,"syntology":null},{"paper":null,"slug":"breaking-the-bank-with-chatgpt-few-shot-text","title":"Breaking the Bank with ChatGPT: Few-Shot Text Classification for Finance","date":"2023-08-28","arxiv_id":"2308.14634","n_code_links":0,"syntology":null},{"paper":"/paper/empowering-clinicians-and-democratizing-data","slug":"empowering-clinicians-and-democratizing-data","title":"Large Language Models Streamline Automated Machine Learning for Clinical Studies","date":"2023-08-27","arxiv_id":"2308.14120","n_code_links":1,"syntology":null},{"paper":"/paper/examining-user-friendly-and-open-sourced","slug":"examining-user-friendly-and-open-sourced","title":"Examining User-Friendly and Open-Sourced Large GPT Models: A Survey on Language, Multimodal, and Scientific GPT Models","date":"2023-08-27","arxiv_id":"2308.14149","n_code_links":1,"syntology":null},{"paper":null,"slug":"medalign-a-clinician-generated-dataset-for","title":"MedAlign: A Clinician-Generated Dataset for Instruction Following with Electronic Medical Records","date":"2023-08-27","arxiv_id":"2308.14089","n_code_links":0,"syntology":null},{"paper":"/paper/a-wide-evaluation-of-chatgpt-on-affective","slug":"a-wide-evaluation-of-chatgpt-on-affective","title":"A Wide Evaluation of ChatGPT on Affective Computing Tasks","date":"2023-08-26","arxiv_id":"2308.13911","n_code_links":1,"syntology":null},{"paper":null,"slug":"adversarial-fine-tuning-of-language-models-an","title":"Adversarial Fine-Tuning of Language Models: An Iterative Optimisation Approach for the Generation and Detection of Problematic Content","date":"2023-08-26","arxiv_id":"2308.13768","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-large-language-models-for-knowledge","slug":"exploring-large-language-models-for-knowledge","title":"Exploring Large Language Models for Knowledge Graph Completion","date":"2023-08-26","arxiv_id":"2308.13916","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yao8839836/kg-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/cultural-alignment-in-large-language-models","slug":"cultural-alignment-in-large-language-models","title":"Cultural Alignment in Large Language Models: An Explanatory Analysis Based on Hofstede's Cultural Dimensions","date":"2023-08-25","arxiv_id":"2309.12342","n_code_links":1,"syntology":null},{"paper":"/paper/do-not-answer-a-dataset-for-evaluating","slug":"do-not-answer-a-dataset-for-evaluating","title":"Do-Not-Answer: A Dataset for Evaluating Safeguards in LLMs","date":"2023-08-25","arxiv_id":"2308.13387","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["libr-ai/do-not-answer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mllm-dataengine-an-iterative-refinement","slug":"mllm-dataengine-an-iterative-refinement","title":"MLLM-DataEngine: An Iterative Refinement Approach for MLLM","date":"2023-08-25","arxiv_id":"2308.13566","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":4,"n_instrument":4,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opendatalab/mllm-dataengine"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"prompting-a-large-language-model-to-generate","title":"Prompting a Large Language Model to Generate Diverse Motivational Messages: A Comparison with Human-Written Messages","date":"2023-08-25","arxiv_id":"2308.13479","n_code_links":0,"syntology":null},{"paper":"/paper/scieval-a-multi-level-large-language-model","slug":"scieval-a-multi-level-large-language-model","title":"SciEval: A Multi-Level Large Language Model Evaluation Benchmark for Scientific Research","date":"2023-08-25","arxiv_id":"2308.13149","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opendfm/bai-scieval","opendfm/scieval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gpteval-a-survey-on-assessments-of-chatgpt","title":"GPTEval: A Survey on Assessments of ChatGPT and GPT-4","date":"2023-08-24","arxiv_id":"2308.12488","n_code_links":0,"syntology":null},{"paper":null,"slug":"harnessing-the-power-of-david-against-goliath","title":"Harnessing the Power of David against Goliath: Exploring Instruction Data Generation without Using Closed-Source Models","date":"2023-08-24","arxiv_id":"2308.12711","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-as-reality-a-co-creative","title":"Language as Reality: A Co-Creative Storytelling Game Experience in 1001 Nights using Generative AI","date":"2023-08-24","arxiv_id":"2308.12915","n_code_links":0,"syntology":null},{"paper":null,"slug":"mind-vs-mouth-on-measuring-re-judge","title":"Mind vs. Mouth: On Measuring Re-judge Inconsistency of Social Bias in Large Language Models","date":"2023-08-24","arxiv_id":"2308.12578","n_code_links":0,"syntology":null},{"paper":"/paper/vigc-visual-instruction-generation-and","slug":"vigc-visual-instruction-generation-and","title":"VIGC: Visual Instruction Generation and Correction","date":"2023-08-24","arxiv_id":"2308.12714","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":1,"n_instrument":5,"unverified":2,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["opendatalab/vigc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"are-chatgpt-and-gpt-4-good-poker-players-a","title":"Are ChatGPT and GPT-4 Good Poker Players? -- A Pre-Flop Analysis","date":"2023-08-23","arxiv_id":"2308.12466","n_code_links":0,"syntology":null},{"paper":"/paper/diagnosing-infeasible-optimization-problems","slug":"diagnosing-infeasible-optimization-problems","title":"Diagnosing Infeasible Optimization Problems Using Large Language Models","date":"2023-08-23","arxiv_id":"2308.12923","n_code_links":1,"syntology":null},{"paper":"/paper/instructiongpt-4-a-200-instruction-paradigm","slug":"instructiongpt-4-a-200-instruction-paradigm","title":"InstructionGPT-4: A 200-Instruction Paradigm for Fine-Tuning MiniGPT-4","date":"2023-08-23","arxiv_id":"2308.12067","n_code_links":3,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["waltonfuture/InstructionGPT-4"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/out-of-the-cage-how-stochastic-parrots-win-in","slug":"out-of-the-cage-how-stochastic-parrots-win-in","title":"Out of the Cage: How Stochastic Parrots Win in Cyber Security Environments","date":"2023-08-23","arxiv_id":"2308.12086","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["stratosphereips/netsecgame"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"prompt-based-length-controlled-generation","title":"Prompt-Based Length Controlled Generation with Reinforcement Learning","date":"2023-08-23","arxiv_id":"2308.12030","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-large-language-models-on-graphs","slug":"evaluating-large-language-models-on-graphs","title":"Evaluating Large Language Models on Graphs: Performance Insights and Comparative Analysis","date":"2023-08-22","arxiv_id":"2308.11224","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ayame1006/llmtograph"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gpt-in-the-loop-adaptive-decision-making-for","title":"GPT-in-the-Loop: Adaptive Decision-Making for Multiagent Systems","date":"2023-08-21","arxiv_id":"2308.10435","n_code_links":0,"syntology":null}],"record_sha256":"77a757503141d9c5a715dcc1be4d058c47d6926ad1b398f56ca77870fdfa1a7f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}