{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/17","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":17,"pages_in_order":29,"rows_per_page":100,"rows":[1601,1700],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/16","next":"/method/gpt-4/papers/18","papers":[{"paper":"/paper/towards-democratized-flood-risk-management-an","slug":"towards-democratized-flood-risk-management-an","title":"Towards Democratized Flood Risk Management: An Advanced AI Assistant Enabled by GPT-4 for Enhanced Interpretability and Public Engagement","date":"2024-03-05","arxiv_id":"2403.03188","n_code_links":2,"syntology":null},{"paper":null,"slug":"towards-training-a-chinese-large-language","title":"Towards Training A Chinese Large Language Model for Anesthesiology","date":"2024-03-05","arxiv_id":"2403.02742","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-generate-architectural-design","title":"Can LLMs Generate Architectural Design Decisions? -An Exploratory Empirical study","date":"2024-03-04","arxiv_id":"2403.01709","n_code_links":0,"syntology":null},{"paper":"/paper/key-point-driven-data-synthesis-with-its","slug":"key-point-driven-data-synthesis-with-its","title":"Key-Point-Driven Data Synthesis with its Enhancement on Mathematical Reasoning","date":"2024-03-04","arxiv_id":"2403.02333","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-oriented-retrieval-tuner","title":"LLM-Oriented Retrieval Tuner","date":"2024-03-04","arxiv_id":"2403.01999","n_code_links":0,"syntology":null},{"paper":null,"slug":"predicting-learning-performance-with-large","title":"Predicting Learning Performance with Large Language Models: A Study in Adult Literacy","date":"2024-03-04","arxiv_id":"2403.14668","n_code_links":0,"syntology":null},{"paper":"/paper/sciassess-benchmarking-llm-proficiency-in","slug":"sciassess-benchmarking-llm-proficiency-in","title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","date":"2024-03-04","arxiv_id":"2403.01976","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sci-assess/sciassess"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/using-llms-for-the-extraction-and","slug":"using-llms-for-the-extraction-and","title":"Using LLMs for the Extraction and Normalization of Product Attribute Values","date":"2024-03-04","arxiv_id":"2403.02130","n_code_links":1,"syntology":null},{"paper":"/paper/varierr-nli-separating-annotation-error-from","slug":"varierr-nli-separating-annotation-error-from","title":"VariErr NLI: Separating Annotation Error from Human Label Variation","date":"2024-03-04","arxiv_id":"2403.01931","n_code_links":0,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":null,"slug":"moviellm-enhancing-long-video-understanding","title":"MovieLLM: Enhancing Long Video Understanding with AI-Generated Movies","date":"2024-03-03","arxiv_id":"2403.01422","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-large-language-models-as-virtual","title":"Evaluating Large Language Models as Virtual Annotators for Time-series Physical Sensing Data","date":"2024-03-02","arxiv_id":"2403.01133","n_code_links":0,"syntology":null},{"paper":"/paper/improving-the-validity-of-automatically","slug":"improving-the-validity-of-automatically","title":"Improving the Validity of Automatically Generated Feedback via Reinforcement Learning","date":"2024-03-02","arxiv_id":"2403.01304","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["umass-ml4ed/feedback-gen-dpo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/lab-large-scale-alignment-for-chatbots","slug":"lab-large-scale-alignment-for-chatbots","title":"LAB: Large-Scale Alignment for ChatBots","date":"2024-03-02","arxiv_id":"2403.01081","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"llamoco-instruction-tuning-of-large-language","title":"LLaMoCo: Instruction Tuning of Large Language Models for Optimization Code Generation","date":"2024-03-02","arxiv_id":"2403.01131","n_code_links":0,"syntology":null},{"paper":null,"slug":"lm4opt-unveiling-the-potential-of-large","title":"LM4OPT: Unveiling the Potential of Large Language Models in Formulating Mathematical Optimization Problems","date":"2024-03-02","arxiv_id":"2403.01342","n_code_links":0,"syntology":null},{"paper":"/paper/reading-subtext-evaluating-large-language","slug":"reading-subtext-evaluating-large-language","title":"Reading Subtext: Evaluating Large Language Models on Short Story Summarization with Writers","date":"2024-03-02","arxiv_id":"2403.01061","n_code_links":2,"syntology":null},{"paper":null,"slug":"a-systematic-evaluation-of-large-language-2","title":"Comparing large language models and human programmers for generating programming code","date":"2024-03-01","arxiv_id":"2403.00894","n_code_links":0,"syntology":null},{"paper":null,"slug":"crimson-empowering-strategic-reasoning-in","title":"Crimson: Empowering Strategic Reasoning in Cybersecurity through Large Language Models","date":"2024-03-01","arxiv_id":"2403.00878","n_code_links":0,"syntology":null},{"paper":null,"slug":"dfin-sql-integrating-focused-schema-with-din","title":"DFIN-SQL: Integrating Focused Schema with DIN-SQL for Superior Accuracy in Large-Scale Databases","date":"2024-03-01","arxiv_id":"2403.00872","n_code_links":0,"syntology":null},{"paper":"/paper/softtiger-a-clinical-foundation-model-for","slug":"softtiger-a-clinical-foundation-model-for","title":"SoftTiger: A Clinical Foundation Model for Healthcare Workflows","date":"2024-03-01","arxiv_id":"2403.00868","n_code_links":1,"syntology":null},{"paper":null,"slug":"surveying-the-dead-minds-historical","title":"Surveying the Dead Minds: Historical-Psychological Text Analysis with Contextualized Construct Representation (CCR) for Classical Chinese","date":"2024-03-01","arxiv_id":"2403.00509","n_code_links":0,"syntology":null},{"paper":null,"slug":"crafting-knowledge-exploring-the-creative","title":"Crafting Knowledge: Exploring the Creative Mechanisms of Chat-Based Search Engines","date":"2024-02-29","arxiv_id":"2402.19421","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-ensemble-optimal-large-language-model","title":"LLM-Ensemble: Optimal Large Language Model Ensemble Method for E-commerce Product Attribute Value Extraction","date":"2024-02-29","arxiv_id":"2403.00863","n_code_links":0,"syntology":null},{"paper":"/paper/newsbench-systematic-evaluation-of-llms-for","slug":"newsbench-systematic-evaluation-of-llms-for","title":"NewsBench: A Systematic Evaluation Framework for Assessing Editorial Capabilities of Large Language Models in Chinese Journalism","date":"2024-02-29","arxiv_id":"2403.00862","n_code_links":1,"syntology":{"ran":0,"of":3,"n_ran_checked":0,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"0 ran · 3 unverified","official":{"repos":["iaar-shanghai/newsbench"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"paper":null,"slug":"on-the-decision-making-abilities-in-role","title":"On the Decision-Making Abilities in Role-Playing using Large Language Models","date":"2024-02-29","arxiv_id":"2402.18807","n_code_links":0,"syntology":null},{"paper":null,"slug":"query-opt-optimizing-inference-of-large","title":"Query-OPT: Optimizing Inference of Large Language Models via Multi-Query Instructions in Meeting Summarization","date":"2024-02-29","arxiv_id":"2403.00067","n_code_links":0,"syntology":null},{"paper":"/paper/teaching-large-language-models-an-unseen","slug":"teaching-large-language-models-an-unseen","title":"Teaching Large Language Models an Unseen Language on the Fly","date":"2024-02-29","arxiv_id":"2402.19167","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["luciusssss/zhuangbench"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"wisdom-of-the-silicon-crowd-llm-ensemble","title":"Wisdom of the Silicon Crowd: LLM Ensemble Prediction Capabilities Rival Human Crowd Accuracy","date":"2024-02-29","arxiv_id":"2402.19379","n_code_links":0,"syntology":null},{"paper":"/paper/x-amr-annotation-tool","slug":"x-amr-annotation-tool","title":"X-AMR Annotation Tool","date":"2024-02-29","arxiv_id":"2403.15407","n_code_links":1,"syntology":null},{"paper":"/paper/clustering-and-ranking-diversity-preserved","slug":"clustering-and-ranking-diversity-preserved","title":"Clustering and Ranking: Diversity-preserved Instruction Selection through Expert-aligned Quality Estimation","date":"2024-02-28","arxiv_id":"2402.18191","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ironbeliever/car"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"few-shot-fairness-unveiling-llm-s-potential","title":"Few-Shot Fairness: Unveiling LLM's Potential for Fairness-Aware Classification","date":"2024-02-28","arxiv_id":"2402.18502","n_code_links":0,"syntology":null},{"paper":"/paper/fofo-a-benchmark-to-evaluate-llms-format","slug":"fofo-a-benchmark-to-evaluate-llms-format","title":"FOFO: A Benchmark to Evaluate LLMs' Format-Following Capability","date":"2024-02-28","arxiv_id":"2402.18667","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["salesforceairesearch/fofo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/hire-a-linguist-learning-endangered-languages","slug":"hire-a-linguist-learning-endangered-languages","title":"Hire a Linguist!: Learning Endangered Languages with In-Context Linguistic Descriptions","date":"2024-02-28","arxiv_id":"2402.18025","n_code_links":2,"syntology":{"ran":8,"of":13,"n_ran_checked":7,"n_instrument":1,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["leililab/lingollm","llilab/llm4endangeredlang"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lemo-nade-multi-parameter-neural-architecture","title":"LeMo-NADe: Multi-Parameter Neural Architecture Discovery with LLMs","date":"2024-02-28","arxiv_id":"2402.18443","n_code_links":0,"syntology":null},{"paper":"/paper/making-them-ask-and-answer-jailbreaking-large","slug":"making-them-ask-and-answer-jailbreaking-large","title":"Making Them Ask and Answer: Jailbreaking Large Language Models in Few Queries via Disguise and Reconstruction","date":"2024-02-28","arxiv_id":"2402.18104","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["llm-dra/dra"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/are-llms-capable-of-data-based-statistical","slug":"are-llms-capable-of-data-based-statistical","title":"Are LLMs Capable of Data-based Statistical and Causal Reasoning? Benchmarking Advanced Quantitative Reasoning with Data","date":"2024-02-27","arxiv_id":"2402.17644","n_code_links":1,"syntology":null},{"paper":null,"slug":"benchmarking-gpt-4-on-algorithmic-problems-a","title":"Benchmarking GPT-4 on Algorithmic Problems: A Systematic Evaluation of Prompting Strategies","date":"2024-02-27","arxiv_id":"2402.17396","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-gpt-4-identify-propaganda-annotation-and","title":"Can GPT-4 Identify Propaganda? Annotation and Detection of Propaganda Spans in News Articles","date":"2024-02-27","arxiv_id":"2402.17478","n_code_links":0,"syntology":null},{"paper":"/paper/can-llm-generate-culturally-relevant","slug":"can-llm-generate-culturally-relevant","title":"Can LLM Generate Culturally Relevant Commonsense QA Data? Case Study in Indonesian and Sundanese","date":"2024-02-27","arxiv_id":"2402.17302","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rifkiaputri/id-csqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ds-agent-automated-data-science-by-empowering","slug":"ds-agent-automated-data-science-by-empowering","title":"DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning","date":"2024-02-27","arxiv_id":"2402.17453","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["guosyjlu/ds-agent"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"omniact-a-dataset-and-benchmark-for-enabling","title":"OmniACT: A Dataset and Benchmark for Enabling Multimodal Generalist Autonomous Agents for Desktop and Web","date":"2024-02-27","arxiv_id":"2402.17553","n_code_links":0,"syntology":null},{"paper":null,"slug":"reasoning-in-conversation-solving-subjective","title":"Reasoning in Conversation: Solving Subjective Tasks through Dialogue Simulation for Large Language Models","date":"2024-02-27","arxiv_id":"2402.17226","n_code_links":0,"syntology":null},{"paper":"/paper/researchy-questions-a-dataset-of-multi","slug":"researchy-questions-a-dataset-of-multi","title":"Researchy Questions: A Dataset of Multi-Perspective, Decompositional Questions for LLM Web Agents","date":"2024-02-27","arxiv_id":"2402.17896","n_code_links":0,"syntology":null},{"paper":"/paper/sofa-shielded-on-the-fly-alignment-via","slug":"sofa-shielded-on-the-fly-alignment-via","title":"SoFA: Shielded On-the-fly Alignment via Priority Rule Following","date":"2024-02-27","arxiv_id":"2402.17358","n_code_links":1,"syntology":null},{"paper":"/paper/songcomposer-a-large-language-model-for-lyric","slug":"songcomposer-a-large-language-model-for-lyric","title":"SongComposer: A Large Language Model for Lyric and Melody Generation in Song Composition","date":"2024-02-27","arxiv_id":"2402.17645","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pjlab-songcomposer/songcomposer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"automated-floodwater-depth-estimation-using","title":"Automated Floodwater Depth Estimation Using Large Multimodal Model for Rapid Flood Mapping","date":"2024-02-26","arxiv_id":"2402.16684","n_code_links":0,"syntology":null},{"paper":"/paper/codes-towards-building-open-source-language","slug":"codes-towards-building-open-source-language","title":"CodeS: Towards Building Open-source Language Models for Text-to-SQL","date":"2024-02-26","arxiv_id":"2402.16347","n_code_links":1,"syntology":{"ran":12,"of":16,"n_ran_checked":12,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ruckbreasoning/codes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generating-effective-ensembles-for-sentiment","title":"Generating Effective Ensembles for Sentiment Analysis","date":"2024-02-26","arxiv_id":"2402.16700","n_code_links":0,"syntology":null},{"paper":null,"slug":"if-in-a-crowdsourced-data-annotation-pipeline","title":"If in a Crowdsourced Data Annotation Pipeline, a GPT-4","date":"2024-02-26","arxiv_id":"2402.16795","n_code_links":0,"syntology":null},{"paper":null,"slug":"mathgenie-generating-synthetic-data-with","title":"MathGenie: Generating Synthetic Data with Question Back-translation for Enhancing Mathematical Reasoning of LLMs","date":"2024-02-26","arxiv_id":"2402.16352","n_code_links":0,"syntology":null},{"paper":null,"slug":"qase-enhanced-plms-improved-control-in-text","title":"QASE Enhanced PLMs: Improved Control in Text Generation for MRC","date":"2024-02-26","arxiv_id":"2403.04771","n_code_links":0,"syntology":null},{"paper":"/paper/chatmusician-understanding-and-generating","slug":"chatmusician-understanding-and-generating","title":"ChatMusician: Understanding and Generating Music Intrinsically with LLM","date":"2024-02-25","arxiv_id":"2402.16153","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hf-lin/ChatMusician"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/drattack-prompt-decomposition-and","slug":"drattack-prompt-decomposition-and","title":"DrAttack: Prompt Decomposition and Reconstruction Makes Powerful LLM Jailbreakers","date":"2024-02-25","arxiv_id":"2402.16914","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["xirui-li/drattack"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/ehrnoteqa-a-patient-specific-question","slug":"ehrnoteqa-a-patient-specific-question","title":"EHRNoteQA: An LLM Benchmark for Real-World Clinical Practice Using Discharge Summaries","date":"2024-02-25","arxiv_id":"2402.16040","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ji-youn-kim/ehrnoteqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/graphwiz-an-instruction-following-language","slug":"graphwiz-an-instruction-following-language","title":"GraphWiz: An Instruction-Following Language Model for Graph Problems","date":"2024-02-25","arxiv_id":"2402.16029","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nuochenpku/Graph-Reasoning-LLM"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-do-humans-write-code-large-models-do-it","slug":"how-do-humans-write-code-large-models-do-it","title":"How Do Humans Write Code? Large Models Do It the Same Way Too","date":"2024-02-24","arxiv_id":"2402.15729","n_code_links":1,"syntology":null},{"paper":"/paper/increasing-sam-zero-shot-performance-on","slug":"increasing-sam-zero-shot-performance-on","title":"TV-SAM: Increasing Zero-Shot Segmentation Performance on Multimodal Medical Images Using GPT-4 Generated Descriptive Prompts Without Human Annotation","date":"2024-02-24","arxiv_id":"2402.15759","n_code_links":1,"syntology":null},{"paper":"/paper/mathwell-generating-educational-math-word","slug":"mathwell-generating-educational-math-word","title":"MATHWELL: Generating Educational Math Word Problems Using Teacher Annotations","date":"2024-02-24","arxiv_id":"2402.15861","n_code_links":2,"syntology":null},{"paper":"/paper/a-data-centric-approach-to-generate-faithful","slug":"a-data-centric-approach-to-generate-faithful","title":"A Data-Centric Approach To Generate Faithful and High Quality Patient Summaries with Large Language Models","date":"2024-02-23","arxiv_id":"2402.15422","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stefanhgm/patient_summaries_with_llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/executing-natural-language-described","slug":"executing-natural-language-described","title":"Executing Natural Language-Described Algorithms with Large Language Models: An Investigation","date":"2024-02-23","arxiv_id":"2403.00795","n_code_links":1,"syntology":null},{"paper":"/paper/how-un-ethical-are-instruction-centric","slug":"how-un-ethical-are-instruction-centric","title":"How (un)ethical are instruction-centric responses of LLMs? Unveiling the vulnerabilities of safety guardrails to harmful queries","date":"2024-02-23","arxiv_id":"2402.15302","n_code_links":1,"syntology":null},{"paper":null,"slug":"interpreting-context-look-ups-in-transformers","title":"Interpreting Context Look-ups in Transformers: Investigating Attention-MLP Interactions","date":"2024-02-23","arxiv_id":"2402.15055","n_code_links":0,"syntology":null},{"paper":"/paper/tombench-benchmarking-theory-of-mind-in-large","slug":"tombench-benchmarking-theory-of-mind-in-large","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","date":"2024-02-23","arxiv_id":"2402.15052","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhchen18/tombench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"can-large-language-models-detect","title":"Can Large Language Models Detect Misinformation in Scientific News Reporting?","date":"2024-02-22","arxiv_id":"2402.14268","n_code_links":0,"syntology":null},{"paper":null,"slug":"copr-continual-human-preference-learning-via","title":"COPR: Continual Human Preference Learning via Optimal Policy Regularization","date":"2024-02-22","arxiv_id":"2402.14228","n_code_links":0,"syntology":null},{"paper":null,"slug":"gate-x-e-a-challenge-set-for-gender-fair","title":"GATE X-E : A Challenge Set for Gender-Fair Translations from Weakly-Gendered Languages","date":"2024-02-22","arxiv_id":"2402.14277","n_code_links":0,"syntology":null},{"paper":"/paper/is-chatgpt-more-empathetic-than-humans","slug":"is-chatgpt-more-empathetic-than-humans","title":"Is ChatGPT More Empathetic than Humans?","date":"2024-02-22","arxiv_id":"2403.05572","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-chatgpt-the-future-of-causal-text-mining-a","title":"Is ChatGPT the Future of Causal Text Mining? A Comprehensive Evaluation and Analysis","date":"2024-02-22","arxiv_id":"2402.14484","n_code_links":0,"syntology":null},{"paper":"/paper/middleware-for-llms-tools-are-instrumental","slug":"middleware-for-llms-tools-are-instrumental","title":"Middleware for LLMs: Tools Are Instrumental for Language Agents in Complex Environments","date":"2024-02-22","arxiv_id":"2402.14672","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["OSU-NLP-Group/Middleware","osu-nlp-group/fuxi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mitigating-fine-tuning-jailbreak-attack-with","slug":"mitigating-fine-tuning-jailbreak-attack-with","title":"Mitigating Fine-tuning based Jailbreak Attack with Backdoor Enhanced Safety Alignment","date":"2024-02-22","arxiv_id":"2402.14968","n_code_links":1,"syntology":null},{"paper":"/paper/opencodeinterpreter-integrating-code","slug":"opencodeinterpreter-integrating-code","title":"OpenCodeInterpreter: Integrating Code Generation with Execution and Refinement","date":"2024-02-22","arxiv_id":"2402.14658","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"roboscript-code-generation-for-free-form","title":"RoboScript: Code Generation for Free-Form Manipulation Tasks across Real and Simulation","date":"2024-02-22","arxiv_id":"2402.14623","n_code_links":0,"syntology":null},{"paper":"/paper/tokenization-counts-the-impact-of","slug":"tokenization-counts-the-impact-of","title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","date":"2024-02-22","arxiv_id":"2402.14903","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aadityasingh/tokenizationcounts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","n_code_links":1,"syntology":{"ran":8,"of":17,"n_ran_checked":8,"n_instrument":0,"unverified":9,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"whose-llm-is-it-anyway-linguistic-comparison","title":"Whose LLM is it Anyway? Linguistic Comparison and LLM Attribution for GPT-3.5, GPT-4 and Bard","date":"2024-02-22","arxiv_id":"2402.14533","n_code_links":0,"syntology":null},{"paper":"/paper/are-llms-effective-negotiators-systematic","slug":"are-llms-effective-negotiators-systematic","title":"Are LLMs Effective Negotiators? Systematic Evaluation of the Multifaceted Capabilities of LLMs in Negotiation Dialogues","date":"2024-02-21","arxiv_id":"2402.13550","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dsincerity/syseval-negollms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"beyond-hate-speech-nlp-s-challenges-and","title":"Beyond Hate Speech: NLP's Challenges and Opportunities in Uncovering Dehumanizing Language","date":"2024-02-21","arxiv_id":"2402.13818","n_code_links":0,"syntology":null},{"paper":"/paper/criticbench-evaluating-large-language-models","slug":"criticbench-evaluating-large-language-models","title":"CriticEval: Evaluating Large Language Model as Critic","date":"2024-02-21","arxiv_id":"2402.13764","n_code_links":2,"syntology":null},{"paper":null,"slug":"data-driven-discovery-with-large-generative","title":"Data-driven Discovery with Large Generative Models","date":"2024-02-21","arxiv_id":"2402.13610","n_code_links":0,"syntology":null},{"paper":null,"slug":"driving-generative-agents-with-their","title":"Driving Generative Agents With Their Personality","date":"2024-02-21","arxiv_id":"2402.14879","n_code_links":0,"syntology":null},{"paper":"/paper/fanoutqa-multi-hop-multi-document-question","slug":"fanoutqa-multi-hop-multi-document-question","title":"FanOutQA: A Multi-Hop, Multi-Document Question Answering Benchmark for Large Language Models","date":"2024-02-21","arxiv_id":"2402.14116","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhudotexe/fanoutqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"kuaiji-the-first-chinese-accounting-large","title":"Kuaiji: the First Chinese Accounting Large Language Model","date":"2024-02-21","arxiv_id":"2402.13866","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-for-data-annotation-a","slug":"large-language-models-for-data-annotation-a","title":"Large Language Models for Data Annotation and Synthesis: A Survey","date":"2024-02-21","arxiv_id":"2402.13446","n_code_links":1,"syntology":null},{"paper":"/paper/omgeval-an-open-multilingual-generative","slug":"omgeval-an-open-multilingual-generative","title":"OMGEval: An Open Multilingual Generative Evaluation Benchmark for Large Language Models","date":"2024-02-21","arxiv_id":"2402.13524","n_code_links":1,"syntology":null},{"paper":"/paper/pca-bench-evaluating-multimodal-large","slug":"pca-bench-evaluating-multimodal-large","title":"PCA-Bench: Evaluating Multimodal Large Language Models in Perception-Cognition-Action Chain","date":"2024-02-21","arxiv_id":"2402.15527","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pkunlp-icler/pca-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/synfac-edit-synthetic-imitation-edit-feedback","slug":"synfac-edit-synthetic-imitation-edit-feedback","title":"SYNFAC-EDIT: Synthetic Imitation Edit Feedback for Factual Alignment in Clinical Summarization","date":"2024-02-21","arxiv_id":"2402.13919","n_code_links":1,"syntology":null},{"paper":null,"slug":"test-driven-development-for-code-generation","title":"Test-Driven Development for Code Generation","date":"2024-02-21","arxiv_id":"2402.13521","n_code_links":0,"syntology":null},{"paper":"/paper/towards-building-multilingual-language-model","slug":"towards-building-multilingual-language-model","title":"Towards Building Multilingual Language Model for Medicine","date":"2024-02-21","arxiv_id":"2402.13963","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["magic-ai4med/mmedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/unigraph-learning-a-cross-domain-graph","slug":"unigraph-learning-a-cross-domain-graph","title":"UniGraph: Learning a Unified Cross-Domain Foundation Model for Text-Attributed Graphs","date":"2024-02-21","arxiv_id":"2402.13630","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":6,"n_instrument":1,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yf-he/UniGraph"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/what-s-in-a-name-auditing-large-language","slug":"what-s-in-a-name-auditing-large-language","title":"What's in a Name? Auditing Large Language Models for Race and Gender Bias","date":"2024-02-21","arxiv_id":"2402.14875","n_code_links":1,"syntology":null},{"paper":null,"slug":"winoviz-probing-visual-properties-of-objects","title":"WinoViz: Probing Visual Properties of Objects Under Different States","date":"2024-02-21","arxiv_id":"2402.13584","n_code_links":0,"syntology":null},{"paper":"/paper/a-survey-on-knowledge-distillation-of-large","slug":"a-survey-on-knowledge-distillation-of-large","title":"A Survey on Knowledge Distillation of Large Language Models","date":"2024-02-20","arxiv_id":"2402.13116","n_code_links":1,"syntology":null},{"paper":null,"slug":"advancing-genai-assisted-programming-a","title":"Advancing GenAI Assisted Programming--A Comparative Study on Prompt Efficiency and Code Quality Between GPT-4 and GLM-4","date":"2024-02-20","arxiv_id":"2402.12782","n_code_links":0,"syntology":null},{"paper":null,"slug":"agentmd-empowering-language-agents-for-risk","title":"AgentMD: Empowering Language Agents for Risk Prediction with Large-Scale Clinical Tool Learning","date":"2024-02-20","arxiv_id":"2402.13225","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-be-used-to-provide","title":"Can Large Language Models be Used to Provide Psychological Counselling? An Analysis of GPT-4-Generated Responses Using Role-play Dialogues","date":"2024-02-20","arxiv_id":"2402.12738","n_code_links":0,"syntology":null},{"paper":"/paper/humaneval-on-latest-gpt-models-2024","slug":"humaneval-on-latest-gpt-models-2024","title":"HumanEval on Latest GPT Models -- 2024","date":"2024-02-20","arxiv_id":"2402.14852","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["daniel442li/gpt-human-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/me-llama-foundation-large-language-models-for","slug":"me-llama-foundation-large-language-models-for","title":"Me LLaMA: Foundation Large Language Models for Medical Applications","date":"2024-02-20","arxiv_id":"2402.12749","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bids-xu-lab/me-llama"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"opdai-at-semeval-2024-task-6-small-llms-can","title":"OPDAI at SemEval-2024 Task 6: Small LLMs can Accelerate Hallucination Detection with Weakly Supervised Data","date":"2024-02-20","arxiv_id":"2402.12913","n_code_links":0,"syntology":null},{"paper":null,"slug":"precise-framework-gpt-based-text-for-improved","title":"PRECISE Framework: GPT-based Text For Improved Readability, Reliability, and Understandability of Radiology Reports For Patient-Centered Care","date":"2024-02-20","arxiv_id":"2403.00788","n_code_links":0,"syntology":null},{"paper":"/paper/the-finben-an-holistic-financial-benchmark","slug":"the-finben-an-holistic-financial-benchmark","title":"FinBen: A Holistic Financial Benchmark for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12659","n_code_links":2,"syntology":{"ran":8,"of":12,"n_ran_checked":6,"n_instrument":2,"unverified":4,"pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["the-finai/pixiu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}}],"record_sha256":"9df8af90542d5800330c9b5e66ac90b88ae8b2009e9909e2d137c2e0598d05b5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}