{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/22","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":22,"pages_in_order":29,"rows_per_page":100,"rows":[2101,2200],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/21","next":"/method/gpt-4/papers/23","papers":[{"paper":null,"slug":"from-text-to-image-exploring-gpt-4vision-s","title":"From Text to Image: Exploring GPT-4Vision's Potential in Advanced Radiological Analysis across Subspecialties","date":"2023-11-24","arxiv_id":"2311.14777","n_code_links":0,"syntology":null},{"paper":null,"slug":"auditing-and-mitigating-cultural-bias-in-llms","title":"Cultural Bias and Cultural Alignment of Large Language Models","date":"2023-11-23","arxiv_id":"2311.14096","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-gpt-4-s-vision-capabilities-on","slug":"evaluating-gpt-4-s-vision-capabilities-on","title":"Evaluating GPT-4's Vision Capabilities on Brazilian University Admission Exams","date":"2023-11-23","arxiv_id":"2311.14169","n_code_links":1,"syntology":null},{"paper":null,"slug":"surpassing-gpt-4-medical-coding-with-a-two","title":"Surpassing GPT-4 Medical Coding with a Two-Stage Approach","date":"2023-11-22","arxiv_id":"2311.13735","n_code_links":0,"syntology":null},{"paper":"/paper/towards-improving-document-understanding-an","slug":"towards-improving-document-understanding-an","title":"Towards Improving Document Understanding: An Exploration on Text-Grounding via MLLMs","date":"2023-11-22","arxiv_id":"2311.13194","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["harrytea/tgdoc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ve-a-chatbot-for-latin","title":"@ve: A Chatbot for Latin","date":"2023-11-22","arxiv_id":"2311.14741","n_code_links":0,"syntology":null},{"paper":null,"slug":"attention-large-multimodal-model-is-watching","title":"GeoLocator: a location-integrated large multimodal model for inferring geo-privacy","date":"2023-11-21","arxiv_id":"2311.13018","n_code_links":0,"syntology":null},{"paper":"/paper/from-classification-to-clinical-insights","slug":"from-classification-to-clinical-insights","title":"From Classification to Clinical Insights: Towards Analyzing and Reasoning About Mobile and Behavioral Health Data With Large Language Models","date":"2023-11-21","arxiv_id":"2311.13063","n_code_links":1,"syntology":null},{"paper":"/paper/gaia-a-benchmark-for-general-ai-assistants","slug":"gaia-a-benchmark-for-general-ai-assistants","title":"GAIA: a benchmark for General AI Assistants","date":"2023-11-21","arxiv_id":"2311.12983","n_code_links":2,"syntology":null},{"paper":null,"slug":"gpt4motion-scripting-physical-motions-in-text","title":"GPT4Motion: Scripting Physical Motions in Text-to-Video Generation via Blender-Oriented GPT Planning","date":"2023-11-21","arxiv_id":"2311.12631","n_code_links":0,"syntology":null},{"paper":"/paper/oasis-data-curation-and-assessment-system-for","slug":"oasis-data-curation-and-assessment-system-for","title":"Oasis: Data Curation and Assessment System for Pretraining of Large Language Models","date":"2023-11-21","arxiv_id":"2311.12537","n_code_links":1,"syntology":null},{"paper":"/paper/evil-geniuses-delving-into-the-safety-of-llm","slug":"evil-geniuses-delving-into-the-safety-of-llm","title":"Evil Geniuses: Delving into the Safety of LLM-based Agents","date":"2023-11-20","arxiv_id":"2311.11855","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["t1ans1r/evil-geniuses"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generating-valid-and-natural-adversarial","title":"Generating Valid and Natural Adversarial Examples with Large Language Models","date":"2023-11-20","arxiv_id":"2311.11861","n_code_links":0,"syntology":null},{"paper":"/paper/gpqa-a-graduate-level-google-proof-q-a","slug":"gpqa-a-graduate-level-google-proof-q-a","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","date":"2023-11-20","arxiv_id":"2311.12022","n_code_links":3,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["idavidrein/gpqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-to-use-large-language-models-for-text","slug":"how-to-use-large-language-models-for-text","title":"Towards Human-Level Text Coding with LLMs: The Case of Fatherhood Roles in Public Policy Documents","date":"2023-11-20","arxiv_id":"2311.11844","n_code_links":1,"syntology":null},{"paper":"/paper/meta-prompting-for-agi-systems","slug":"meta-prompting-for-agi-systems","title":"Meta Prompting for AI Systems","date":"2023-11-20","arxiv_id":"2311.11482","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["meta-prompting/meta-prompting"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/which-ai-technique-is-better-to-classify","slug":"which-ai-technique-is-better-to-classify","title":"Which AI Technique Is Better to Classify Requirements? An Experiment with SVM, LSTM, and ChatGPT","date":"2023-11-20","arxiv_id":"2311.11547","n_code_links":1,"syntology":null},{"paper":null,"slug":"behavior-optimized-image-generation","title":"Behavior Optimized Image Generation","date":"2023-11-18","arxiv_id":"2311.10995","n_code_links":0,"syntology":null},{"paper":null,"slug":"visual-ai-and-linguistic-intelligence-through","title":"Visual AI and Linguistic Intelligence Through Steerability and Composability","date":"2023-11-18","arxiv_id":"2312.12383","n_code_links":0,"syntology":null},{"paper":null,"slug":"eduquick-a-dataset-toward-evaluating","title":"EduQuick: A Dataset Toward Evaluating Summarization of Informal Educational Content for Social Media","date":"2023-11-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/taco-enhancing-cross-lingual-transfer-for-low","slug":"taco-enhancing-cross-lingual-transfer-for-low","title":"TaCo: Enhancing Cross-Lingual Transfer for Low-Resource Languages in LLMs through Translation-Assisted Chain-of-Thought Processes","date":"2023-11-17","arxiv_id":"2311.10797","n_code_links":1,"syntology":null},{"paper":"/paper/blt-can-large-language-models-handle-basic","slug":"blt-can-large-language-models-handle-basic","title":"BLT: Can Large Language Models Handle Basic Legal Text?","date":"2023-11-16","arxiv_id":"2311.09693","n_code_links":1,"syntology":null},{"paper":null,"slug":"enchancing-semi-supervised-learning-for","title":"Prompt-based Pseudo-labeling Strategy for Sample-Efficient Semi-Supervised Extractive Summarization","date":"2023-11-16","arxiv_id":"2311.09559","n_code_links":0,"syntology":null},{"paper":"/paper/gee-grammar-error-explanation-with-large","slug":"gee-grammar-error-explanation-with-large","title":"GEE! Grammar Error Explanation with Large Language Models","date":"2023-11-16","arxiv_id":"2311.09517","n_code_links":1,"syntology":null},{"paper":"/paper/huatuogpt-ii-one-stage-training-for-medical","slug":"huatuogpt-ii-one-stage-training-for-medical","title":"HuatuoGPT-II, One-stage Training for Medical Adaption of LLMs","date":"2023-11-16","arxiv_id":"2311.09774","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":11,"n_instrument":2,"unverified":1,"pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["freedomintelligence/huatuogpt-ii"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"human-still-wins-over-llm-an-empirical-study","title":"Human Still Wins over LLM: An Empirical Study of Active Learning on Domain-Specific Annotation Tasks","date":"2023-11-16","arxiv_id":"2311.09825","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigating-data-contamination-in-modern","title":"Investigating Data Contamination in Modern Benchmarks for Large Language Models","date":"2023-11-16","arxiv_id":"2311.09783","n_code_links":0,"syntology":null},{"paper":"/paper/knowledgemath-knowledge-intensive-math-word","slug":"knowledgemath-knowledge-intensive-math-word","title":"FinanceMath: Knowledge-Intensive Math Reasoning in Finance Domains","date":"2023-11-16","arxiv_id":"2311.09797","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":12,"n_instrument":0,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yale-nlp/knowledgemath"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-for-propaganda-span","slug":"large-language-models-for-propaganda-span","title":"Large Language Models for Propaganda Span Annotation","date":"2023-11-16","arxiv_id":"2311.09812","n_code_links":1,"syntology":null},{"paper":"/paper/ml-bench-large-language-models-leverage-open","slug":"ml-bench-large-language-models-leverage-open","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","date":"2023-11-16","arxiv_id":"2311.09835","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gersteinlab/ml-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-evaluating-the-integration-of-reasoning","title":"On Evaluating the Integration of Reasoning and Action in LLM Agents with Database Question Answering","date":"2023-11-16","arxiv_id":"2311.09721","n_code_links":0,"syntology":null},{"paper":null,"slug":"psybench-a-balanced-and-in-depth","title":"ConceptPsy:A Benchmark Suite with Conceptual Comprehensiveness in Psychology","date":"2023-11-16","arxiv_id":"2311.09861","n_code_links":0,"syntology":null},{"paper":"/paper/score-a-framework-for-self-contradictory","slug":"score-a-framework-for-self-contradictory","title":"Self-Contradictory Reasoning Evaluation and Detection","date":"2023-11-16","arxiv_id":"2311.09603","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["uscnlp-lime/Self-Contradictory"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/structured-chemistry-reasoning-with-large","slug":"structured-chemistry-reasoning-with-large","title":"Structured Chemistry Reasoning with Large Language Models","date":"2023-11-16","arxiv_id":"2311.09656","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ozyyshr/structchem"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-autonomous-hypothesis-verification","title":"Towards Autonomous Hypothesis Verification via Language Models with Minimal Guidance","date":"2023-11-16","arxiv_id":"2311.09706","n_code_links":0,"syntology":null},{"paper":"/paper/unifiedvisiongpt-streamlining-vision-oriented","slug":"unifiedvisiongpt-streamlining-vision-oriented","title":"UnifiedVisionGPT: Streamlining Vision-Oriented AI through Generalized Multimodal Framework","date":"2023-11-16","arxiv_id":"2311.10125","n_code_links":1,"syntology":null},{"paper":"/paper/can-large-language-models-follow-concept","slug":"can-large-language-models-follow-concept","title":"Can Large Language Models Follow Concept Annotation Guidelines? A Case Study on Scientific and Financial Domains","date":"2023-11-15","arxiv_id":"2311.08704","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-machine-translation-through","title":"Enhancing Machine Translation through Advanced In-Context Learning: A Methodological Strategy for GPT-4 Improvement","date":"2023-11-15","arxiv_id":"2311.10765","n_code_links":0,"syntology":null},{"paper":"/paper/factcheck-gpt-end-to-end-fine-grained","slug":"factcheck-gpt-end-to-end-fine-grained","title":"Factcheck-Bench: Fine-Grained Evaluation Benchmark for Automatic Fact-checkers","date":"2023-11-15","arxiv_id":"2311.09000","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuxiaw/factcheck-gpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"grim-graph-based-interactive-narrative","title":"GENEVA: GENErating and Visualizing branching narratives using LLMs","date":"2023-11-15","arxiv_id":"2311.09213","n_code_links":0,"syntology":null},{"paper":"/paper/i-was-blind-but-now-i-see-implementing-vision","slug":"i-was-blind-but-now-i-see-implementing-vision","title":"I Was Blind but Now I See: Implementing Vision-Enabled Dialogue in Social Robots","date":"2023-11-15","arxiv_id":"2311.08957","n_code_links":1,"syntology":null},{"paper":null,"slug":"jailbreaking-gpt-4v-via-self-adversarial","title":"Jailbreaking GPT-4V via Self-Adversarial Attacks with System Prompts","date":"2023-11-15","arxiv_id":"2311.09127","n_code_links":0,"syntology":null},{"paper":null,"slug":"llamas-know-what-gpts-don-t-show-surrogate","title":"Llamas Know What GPTs Don't Show: Surrogate Models for Confidence Estimation","date":"2023-11-15","arxiv_id":"2311.08877","n_code_links":0,"syntology":null},{"paper":"/paper/mela-multilingual-evaluation-of-linguistic","slug":"mela-multilingual-evaluation-of-linguistic","title":"MELA: Multilingual Evaluation of Linguistic Acceptability","date":"2023-11-15","arxiv_id":"2311.09033","n_code_links":1,"syntology":null},{"paper":"/paper/safer-instruct-aligning-language-models-with","slug":"safer-instruct-aligning-language-models-with","title":"Safer-Instruct: Aligning Language Models with Automated Preference Data","date":"2023-11-15","arxiv_id":"2311.08685","n_code_links":1,"syntology":null},{"paper":"/paper/tooltalk-evaluating-tool-usage-in-a","slug":"tooltalk-evaluating-tool-usage-in-a","title":"ToolTalk: Evaluating Tool-Usage in a Conversational Setting","date":"2023-11-15","arxiv_id":"2311.10775","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"x-eval-generalizable-multi-aspect-text","title":"X-Eval: Generalizable Multi-aspect Text Evaluation via Augmented Instruction Tuning with Auxiliary Evaluation Aspects","date":"2023-11-15","arxiv_id":"2311.08788","n_code_links":0,"syntology":null},{"paper":"/paper/a-wolf-in-sheep-s-clothing-generalized-nested","slug":"a-wolf-in-sheep-s-clothing-generalized-nested","title":"A Wolf in Sheep's Clothing: Generalized Nested Jailbreak Prompts can Fool Large Language Models Easily","date":"2023-11-14","arxiv_id":"2311.08268","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":7,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["NJUNLP/ReNeLLM"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/automated-title-and-abstract-screening-for","slug":"automated-title-and-abstract-screening-for","title":"Automated title and abstract screening for scoping reviews using the GPT-4 Large Language Model","date":"2023-11-14","arxiv_id":"2311.07918","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparing-humans-gpt-4-and-gpt-4v-on","title":"Comparing Humans, GPT-4, and GPT-4V On Abstraction and Reasoning Tasks","date":"2023-11-14","arxiv_id":"2311.09247","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-llms-on-document-based-qa-exact","title":"Evaluating LLMs on Document-Based QA: Exact Answer Selection and Numerical Extraction using Cogtale dataset","date":"2023-11-14","arxiv_id":"2311.07878","n_code_links":0,"syntology":null},{"paper":null,"slug":"how-good-are-large-language-models-on-african","title":"How good are Large Language Models on African Languages?","date":"2023-11-14","arxiv_id":"2311.07978","n_code_links":0,"syntology":null},{"paper":"/paper/magic-benchmarking-large-language-model","slug":"magic-benchmarking-large-language-model","title":"MAgIC: Investigation of Large Language Model Powered Multi-Agent in Cognition, Adaptability, Rationality and Collaboration","date":"2023-11-14","arxiv_id":"2311.08562","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cathyxl/magic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"simplesafetytests-a-test-suite-for","title":"SimpleSafetyTests: a Test Suite for Identifying Critical Safety Risks in Large Language Models","date":"2023-11-14","arxiv_id":"2311.08370","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-benchmark-to-understand-the-role-of","title":"A Benchmark to Understand the Role of Knowledge Graphs on Large Language Model's Accuracy for Question Answering on Enterprise SQL Databases","date":"2023-11-13","arxiv_id":"2311.07509","n_code_links":0,"syntology":null},{"paper":"/paper/assessing-logical-puzzle-solving-in-large","slug":"assessing-logical-puzzle-solving-in-large","title":"Assessing Logical Puzzle Solving in Large Language Models: Insights from a Minesweeper Case Study","date":"2023-11-13","arxiv_id":"2311.07387","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yinghao-li/minesweeper-for-llm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lm-polygraph-uncertainty-estimation-for","title":"LM-Polygraph: Uncertainty Estimation for Language Models","date":"2023-11-13","arxiv_id":"2311.07383","n_code_links":0,"syntology":null},{"paper":null,"slug":"megaverse-benchmarking-large-language-models","title":"MEGAVERSE: Benchmarking Large Language Models Across Languages, Modalities, Models and Tasks","date":"2023-11-13","arxiv_id":"2311.07463","n_code_links":0,"syntology":null},{"paper":null,"slug":"speech-based-slot-filling-using-large","title":"Speech-based Slot Filling using Large Language Models","date":"2023-11-13","arxiv_id":"2311.07418","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-impact-of-large-language-models-on","title":"The Impact of Large Language Models on Scientific Discovery: a Preliminary Study using GPT-4","date":"2023-11-13","arxiv_id":"2311.07361","n_code_links":0,"syntology":null},{"paper":"/paper/veritymath-advancing-mathematical-reasoning","slug":"veritymath-advancing-mathematical-reasoning","title":"VerityMath: Advancing Mathematical Reasoning by Self-Verification Through Unit Consistency","date":"2023-11-13","arxiv_id":"2311.07172","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vernontoh/veritymath"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"detecting-and-correcting-hate-speech-in","title":"Detecting and Correcting Hate Speech in Multimodal Memes with Large Visual Language Model","date":"2023-11-12","arxiv_id":"2311.06737","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-gpt-4-for-chest-x-ray","title":"Evaluation of GPT-4 for chest X-ray impression generation: A reader study on performance and perception","date":"2023-11-12","arxiv_id":"2311.06815","n_code_links":0,"syntology":null},{"paper":"/paper/flames-benchmarking-value-alignment-of","slug":"flames-benchmarking-value-alignment-of","title":"Flames: Benchmarking Value Alignment of LLMs in Chinese","date":"2023-11-12","arxiv_id":"2311.06899","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-understanding-of-math","title":"Large Language Models' Understanding of Math: Source Criticism and Extrapolation","date":"2023-11-12","arxiv_id":"2311.07618","n_code_links":0,"syntology":null},{"paper":null,"slug":"intentional-biases-in-llm-responses","title":"Intentional Biases in LLM Responses","date":"2023-11-11","arxiv_id":"2311.07611","n_code_links":0,"syntology":null},{"paper":"/paper/data-contamination-quiz-a-tool-to-detect-and","slug":"data-contamination-quiz-a-tool-to-detect-and","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","date":"2023-11-10","arxiv_id":"2311.06233","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shahriargolchin/dcq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-to-bridge-the-gap-between-modalities-a","title":"How to Bridge the Gap between Modalities: Survey on Multimodal Large Language Model","date":"2023-11-10","arxiv_id":"2311.07594","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-models-can-be-logical-solvers","title":"Language Models can be Logical Solvers","date":"2023-11-10","arxiv_id":"2311.06158","n_code_links":0,"syntology":null},{"paper":null,"slug":"making-llms-worth-every-penny-resource","title":"Making LLMs Worth Every Penny: Resource-Limited Text Classification in Banking","date":"2023-11-10","arxiv_id":"2311.06102","n_code_links":0,"syntology":null},{"paper":"/paper/conic10k-a-challenging-math-problem","slug":"conic10k-a-challenging-math-problem","title":"Conic10K: A Challenging Math Problem Understanding and Reasoning Dataset","date":"2023-11-09","arxiv_id":"2311.05113","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-personality-tests-generalize-to-large","title":"Challenging the Validity of Personality Tests for Large Language Models","date":"2023-11-09","arxiv_id":"2311.05297","n_code_links":0,"syntology":null},{"paper":"/paper/technical-report-large-language-models-can","slug":"technical-report-large-language-models-can","title":"Large Language Models can Strategically Deceive their Users when Put Under Pressure","date":"2023-11-09","arxiv_id":"2311.07590","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["apolloresearch/insider-trading"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rethinking-benchmark-and-contamination-for","slug":"rethinking-benchmark-and-contamination-for","title":"Rethinking Benchmark and Contamination for Language Models with Rephrased Samples","date":"2023-11-08","arxiv_id":"2311.04850","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lm-sys/llm-decontaminator"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/black-box-prompt-optimization-aligning-large","slug":"black-box-prompt-optimization-aligning-large","title":"Black-Box Prompt Optimization: Aligning Large Language Models without Model Training","date":"2023-11-07","arxiv_id":"2311.04155","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["thu-coai/bpo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-large-language-models-in","title":"Evaluating Large Language Models in Ophthalmology","date":"2023-11-07","arxiv_id":"2311.04933","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-multiple-large-language-models-in","title":"Evaluating multiple large language models in pediatric ophthalmology","date":"2023-11-07","arxiv_id":"2311.04368","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-and-mitigating-vulnerabilities-in","title":"Identifying and Mitigating Vulnerabilities in LLM-Integrated Applications","date":"2023-11-07","arxiv_id":"2311.16153","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-large-language-models-for-3","title":"Leveraging Large Language Models for Automated Proof Synthesis in Rust","date":"2023-11-07","arxiv_id":"2311.03739","n_code_links":0,"syntology":null},{"paper":"/paper/which-is-better-exploring-prompting-strategy","slug":"which-is-better-exploring-prompting-strategy","title":"Which is better? Exploring Prompting Strategy For LLM-based Metrics","date":"2023-11-07","arxiv_id":"2311.03754","n_code_links":1,"syntology":null},{"paper":"/paper/can-llms-follow-simple-rules","slug":"can-llms-follow-simple-rules","title":"Can LLMs Follow Simple Rules?","date":"2023-11-06","arxiv_id":"2311.04235","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["normster/llm_rules"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/deepinception-hypnotize-large-language-model","slug":"deepinception-hypnotize-large-language-model","title":"DeepInception: Hypnotize Large Language Model to Be Jailbreaker","date":"2023-11-06","arxiv_id":"2311.03191","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tmlr-group/deepinception"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"nexus-at-araieval-shared-task-fine-tuning","title":"Nexus at ArAIEval Shared Task: Fine-Tuning Arabic Language Models for Propaganda and Disinformation Detection","date":"2023-11-06","arxiv_id":"2311.03184","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-and-transferable-black-box","title":"Scalable and Transferable Black-Box Jailbreaks for Language Models via Persona Modulation","date":"2023-11-06","arxiv_id":"2311.03348","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-potential-of-leading-large","title":"Evaluating the Potential of Leading Large Language Models in Reasoning Biology Questions","date":"2023-11-05","arxiv_id":"2311.07582","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-grounding-potential-of-vqa-oriented","slug":"exploring-grounding-potential-of-vqa-oriented","title":"GPT-4V-AD: Exploring Grounding Potential of VQA-oriented GPT-4V for Zero-shot Anomaly Detection","date":"2023-11-05","arxiv_id":"2311.02612","n_code_links":1,"syntology":null},{"paper":null,"slug":"floodbrain-flood-disaster-reporting-by-web","title":"FloodBrain: Flood Disaster Reporting by Web-based Retrieval Augmented Generation with an LLM","date":"2023-11-05","arxiv_id":"2311.02597","n_code_links":0,"syntology":null},{"paper":"/paper/mftcoder-boosting-code-llms-with-multitask","slug":"mftcoder-boosting-code-llms-with-multitask","title":"MFTCoder: Boosting Code LLMs with Multitask Fine-Tuning","date":"2023-11-04","arxiv_id":"2311.02303","n_code_links":1,"syntology":null},{"paper":"/paper/dialogbench-evaluating-llms-as-human-like","slug":"dialogbench-evaluating-llms-as-human-like","title":"DialogBench: Evaluating LLMs as Human-like Dialogue Systems","date":"2023-11-03","arxiv_id":"2311.01677","n_code_links":1,"syntology":null},{"paper":"/paper/pptc-benchmark-evaluating-large-language","slug":"pptc-benchmark-evaluating-large-language","title":"PPTC Benchmark: Evaluating Large Language Models for PowerPoint Task Completion","date":"2023-11-03","arxiv_id":"2311.01767","n_code_links":1,"syntology":{"ran":13,"of":17,"n_ran_checked":13,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["gydpku/pptc"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-risks-of-risk-based-ai-regulation-taking","title":"The risks of risk-based AI regulation: taking liability seriously","date":"2023-11-03","arxiv_id":"2311.14684","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-input-towards-next-generation","title":"Generative Input: Towards Next-Generation Input Methods Paradigm","date":"2023-11-02","arxiv_id":"2311.01166","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-large-language-models-reliable-judges-a","title":"Are Large Language Models Reliable Judges? A Study on the Factuality Evaluation Capabilities of LLMs","date":"2023-11-01","arxiv_id":"2311.00681","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-capture-public","title":"Can Large Language Models Capture Public Opinion about Global Warming? An Empirical Assessment of Algorithmic Fidelity and Bias","date":"2023-11-01","arxiv_id":"2311.00217","n_code_links":0,"syntology":null},{"paper":"/paper/from-text-to-structure-using-large-language","slug":"from-text-to-structure-using-large-language","title":"From Text to Structure: Using Large Language Models to Support the Development of Legal Expert Systems","date":"2023-11-01","arxiv_id":"2311.04911","n_code_links":1,"syntology":null},{"paper":null,"slug":"chipnemo-domain-adapted-llms-for-chip-design","title":"ChipNeMo: Domain-Adapted LLMs for Chip Design","date":"2023-10-31","arxiv_id":"2311.00176","n_code_links":0,"syntology":null},{"paper":null,"slug":"does-gpt-4-pass-the-turing-test","title":"Does GPT-4 pass the Turing test?","date":"2023-10-31","arxiv_id":"2310.20216","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-classification-of-student-help","title":"Efficient Classification of Student Help Requests in Programming Courses Using Large Language Models","date":"2023-10-31","arxiv_id":"2310.20105","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-mistakes-makes-llm-better","slug":"learning-from-mistakes-makes-llm-better","title":"Learning From Mistakes Makes LLM Better Reasoner","date":"2023-10-31","arxiv_id":"2310.20689","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/lema"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bioinstruct-instruction-tuning-of-large","title":"BioInstruct: Instruction Tuning of Large Language Models for Biomedical Natural Language Processing","date":"2023-10-30","arxiv_id":"2310.19975","n_code_links":0,"syntology":null}],"record_sha256":"a1e6a41be447224444a1c2d9ebe2148796e306e7ba1fa835113871a8e60840a4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}