{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/12","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":12,"pages_in_order":29,"rows_per_page":100,"rows":[1101,1200],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/11","next":"/method/gpt-4/papers/13","papers":[{"paper":null,"slug":"grading-massive-open-online-courses-using","title":"Grading Massive Open Online Courses Using Large Language Models","date":"2024-06-16","arxiv_id":"2406.11102","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-synthetic-logical-reasoning-datasets","slug":"scaling-synthetic-logical-reasoning-datasets","title":"Scaling Synthetic Logical Reasoning Datasets with Context-Sensitive Declarative Grammars","date":"2024-06-16","arxiv_id":"2406.11035","n_code_links":1,"syntology":null},{"paper":null,"slug":"wildvision-evaluating-vision-language-models","title":"WildVision: Evaluating Vision-Language Models in the Wild with Human Preferences","date":"2024-06-16","arxiv_id":"2406.11069","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-raw-videos-understanding-edited-videos","slug":"beyond-raw-videos-understanding-edited-videos","title":"Beyond Raw Videos: Understanding Edited Videos with Large Multimodal Model","date":"2024-06-15","arxiv_id":"2406.10484","n_code_links":1,"syntology":null},{"paper":null,"slug":"bridging-the-gap-in-drug-safety-data-analysis","title":"Automating Pharmacovigilance Evidence Generation: Using Large Language Models to Produce Context-Aware SQL","date":"2024-06-15","arxiv_id":"2406.10690","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparsecl-sparse-contrastive-learning-for","title":"SparseCL: Sparse Contrastive Learning for Contradiction Retrieval","date":"2024-06-15","arxiv_id":"2406.10746","n_code_links":0,"syntology":null},{"paper":"/paper/blend-a-benchmark-for-llms-on-everyday","slug":"blend-a-benchmark-for-llms-on-everyday","title":"BLEnD: A Benchmark for LLMs on Everyday Knowledge in Diverse Cultures and Languages","date":"2024-06-14","arxiv_id":"2406.09948","n_code_links":1,"syntology":{"ran":1,"of":10,"n_ran_checked":0,"n_instrument":1,"unverified":9,"pointer_only":10,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["nlee0212/blend"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/bootstrapping-language-models-with-dpo","slug":"bootstrapping-language-models-with-dpo","title":"Bootstrapping Language Models with DPO Implicit Rewards","date":"2024-06-14","arxiv_id":"2406.09760","n_code_links":1,"syntology":{"ran":0,"of":2,"n_ran_checked":0,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"0 ran · 2 unverified","official":{"repos":["sail-sg/dice"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"paper":null,"slug":"domain-specific-shorthand-for-generation","title":"Domain-Specific Shorthand for Generation Based on Context-Free Grammar","date":"2024-06-14","arxiv_id":"2406.10442","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-llm-driven-user-intent","title":"Evaluating LLM-driven User-Intent Formalization for Verification-Aware Languages","date":"2024-06-14","arxiv_id":"2406.09757","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-4o-visual-perception-performance-of","slug":"gpt-4o-visual-perception-performance-of","title":"GPT-4o: Visual perception performance of multimodal large language models in piglet activity understanding","date":"2024-06-14","arxiv_id":"2406.09781","n_code_links":0,"syntology":null},{"paper":"/paper/know-the-unknown-an-uncertainty-sensitive","slug":"know-the-unknown-an-uncertainty-sensitive","title":"Know the Unknown: An Uncertainty-Sensitive Method for LLM Instruction Tuning","date":"2024-06-14","arxiv_id":"2406.10099","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jiaqili404/trustworthyrag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/neural-concept-binder","slug":"neural-concept-binder","title":"Neural Concept Binder","date":"2024-06-14","arxiv_id":"2406.09949","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-research/neuralconceptbinder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/jailbreakeval-an-integrated-toolkit-for","slug":"jailbreakeval-an-integrated-toolkit-for","title":"JailbreakEval: An Integrated Toolkit for Evaluating Jailbreak Attempts Against Large Language Models","date":"2024-06-13","arxiv_id":"2406.09321","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thuccslab/jailbreakeval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"linguistic-bias-in-chatgpt-language-models","title":"Linguistic Bias in ChatGPT: Language Models Reinforce Dialect Discrimination","date":"2024-06-13","arxiv_id":"2406.08818","n_code_links":0,"syntology":null},{"paper":null,"slug":"omnih2o-universal-and-dexterous-human-to","title":"OmniH2O: Universal and Dexterous Human-to-Humanoid Whole-Body Teleoperation and Learning","date":"2024-06-13","arxiv_id":"2406.08858","n_code_links":0,"syntology":null},{"paper":null,"slug":"readctrl-personalizing-text-generation-with","title":"ReadCtrl: Personalizing text generation with readability-controlled instruction learning","date":"2024-06-13","arxiv_id":"2406.09205","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-sociotechnical-lens-for-evaluating-computer","title":"A Sociotechnical Lens for Evaluating Computer Vision Models: A Case Study on Detecting and Reasoning about Gender and Emotion","date":"2024-06-12","arxiv_id":"2406.08222","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-information-extraction-from-thyroid","title":"Automated Information Extraction from Thyroid Operation Narrative: A Comparative Study of GPT-4 and Fine-tuned KoELECTRA","date":"2024-06-12","arxiv_id":"2406.07922","n_code_links":0,"syntology":null},{"paper":"/paper/dafnybench-a-benchmark-for-formal-software","slug":"dafnybench-a-benchmark-for-formal-software","title":"DafnyBench: A Benchmark for Formal Software Verification","date":"2024-06-12","arxiv_id":"2406.08467","n_code_links":1,"syntology":null},{"paper":"/paper/helpsteer2-open-source-dataset-for-training","slug":"helpsteer2-open-source-dataset-for-training","title":"HelpSteer2: Open-source dataset for training top-performing reward models","date":"2024-06-12","arxiv_id":"2406.08673","n_code_links":1,"syntology":null},{"paper":null,"slug":"how-well-it-works-benchmarking-performance-of","title":"How well it works: Benchmarking performance of GPT models on medical natural language processing tasks","date":"2024-06-12","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/indirectrequests-making-task-oriented","slug":"indirectrequests-making-task-oriented","title":"Making Task-Oriented Dialogue Datasets More Natural by Synthetically Generating Indirect User Requests","date":"2024-06-12","arxiv_id":"2406.07794","n_code_links":0,"syntology":null},{"paper":"/paper/judging-the-judges-a-systematic-investigation","slug":"judging-the-judges-a-systematic-investigation","title":"Judging the Judges: A Systematic Study of Position Bias in LLM-as-a-Judge","date":"2024-06-12","arxiv_id":"2406.07791","n_code_links":1,"syntology":{"ran":16,"of":16,"n_ran_checked":16,"n_instrument":0,"unverified":0,"pointer_only":16,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Slimshilin/Position-Bias-Analyzer-Demo"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mistral-c2f-coarse-to-fine-actor-for","title":"Mistral-C2F: Coarse to Fine Actor for Analytical and Reasoning Enhancement in RLHF and Effective-Merged LLMs","date":"2024-06-12","arxiv_id":"2406.08657","n_code_links":0,"syntology":null},{"paper":null,"slug":"supportiveness-based-knowledge-rewriting-for","title":"Supportiveness-based Knowledge Rewriting for Retrieval-augmented Language Modeling","date":"2024-06-12","arxiv_id":"2406.08116","n_code_links":0,"syntology":null},{"paper":"/paper/tailoring-generative-ai-chatbots-for","slug":"tailoring-generative-ai-chatbots-for","title":"Tailoring Generative AI Chatbots for Multiethnic Communities in Disaster Preparedness Communication: Extending the CASA Paradigm","date":"2024-06-12","arxiv_id":"2406.08411","n_code_links":1,"syntology":null},{"paper":null,"slug":"veract-scan-retrieval-augmented-fake-news","title":"VeraCT Scan: Retrieval-Augmented Fake News Detection with Justifiable Reasoning","date":"2024-06-12","arxiv_id":"2406.10289","n_code_links":0,"syntology":null},{"paper":"/paper/what-if-we-recaption-billions-of-web-images","slug":"what-if-we-recaption-billions-of-web-images","title":"What If We Recaption Billions of Web Images with LLaMA-3?","date":"2024-06-12","arxiv_id":"2406.08478","n_code_links":0,"syntology":null},{"paper":"/paper/ai-sandbagging-language-models-can","slug":"ai-sandbagging-language-models-can","title":"AI Sandbagging: Language Models can Strategically Underperform on Evaluations","date":"2024-06-11","arxiv_id":"2406.07358","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["teunvdweij/sandbagging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"beyond-words-on-large-language-models","title":"Beyond Words: On Large Language Models Actionability in Mission-Critical Risk Analysis","date":"2024-06-11","arxiv_id":"2406.10273","n_code_links":0,"syntology":null},{"paper":"/paper/dara-decomposition-alignment-reasoning","slug":"dara-decomposition-alignment-reasoning","title":"DARA: Decomposition-Alignment-Reasoning Autonomous Language Agent for Question Answering over Knowledge Graphs","date":"2024-06-11","arxiv_id":"2406.07080","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["UKPLab/acl2024-DARA"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mllmguard-a-multi-dimensional-safety","slug":"mllmguard-a-multi-dimensional-safety","title":"MLLMGuard: A Multi-dimensional Safety Evaluation Suite for Multimodal Large Language Models","date":"2024-06-11","arxiv_id":"2406.07594","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Carol-gutianle/MLLMGuard"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"validating-llm-generated-programs-with","title":"Validating LLM-Generated Programs with Metamorphic Prompt Testing","date":"2024-06-11","arxiv_id":"2406.06864","n_code_links":0,"syntology":null},{"paper":null,"slug":"annotation-alignment-comparing-llm-and-human","title":"Annotation alignment: Comparing LLM and human annotations of conversational safety","date":"2024-06-10","arxiv_id":"2406.06369","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-language-models-serve-as-text-based-world","title":"Can Language Models Serve as Text-Based World Simulators?","date":"2024-06-10","arxiv_id":"2406.06485","n_code_links":0,"syntology":null},{"paper":"/paper/data-efficient-learning-with-neural-programs","slug":"data-efficient-learning-with-neural-programs","title":"Data-Efficient Learning with Neural Programs","date":"2024-06-10","arxiv_id":"2406.06246","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["alaiasolkobreslin/ised"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/husky-a-unified-open-source-language-agent","slug":"husky-a-unified-open-source-language-agent","title":"Husky: A Unified, Open-Source Language Agent for Multi-Step Reasoning","date":"2024-06-10","arxiv_id":"2406.06469","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["agent-husky/husky-v1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/in-context-learning-and-fine-tuning-gpt-for","slug":"in-context-learning-and-fine-tuning-gpt-for","title":"In-Context Learning and Fine-Tuning GPT for Argument Mining","date":"2024-06-10","arxiv_id":"2406.06699","n_code_links":1,"syntology":null},{"paper":null,"slug":"securenet-a-comparative-study-of-deberta-and","title":"SecureNet: A Comparative Study of DeBERTa and Large Language Models for Phishing Detection","date":"2024-06-10","arxiv_id":"2406.06663","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-knowledge-component-based-methodology-for","title":"A Knowledge-Component-Based Methodology for Evaluating AI Assistants","date":"2024-06-09","arxiv_id":"2406.05603","n_code_links":0,"syntology":null},{"paper":"/paper/are-large-language-models-actually-good-at","slug":"are-large-language-models-actually-good-at","title":"Are Large Language Models Actually Good at Text Style Transfer?","date":"2024-06-09","arxiv_id":"2406.05885","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-efficacy-of-large-language-1","title":"Exploring the Efficacy of Large Language Models (GPT-4) in Binary Reverse Engineering","date":"2024-06-09","arxiv_id":"2406.06637","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-memorize-sensor","title":"Large Language Models Memorize Sensor Datasets! Implications on Human Activity Recognition Research","date":"2024-06-09","arxiv_id":"2406.05900","n_code_links":0,"syntology":null},{"paper":null,"slug":"text2vp-generative-ai-for-visual-programming","title":"Text2VP: Generative AI for Visual Programming and Parametric Modeling","date":"2024-06-09","arxiv_id":"2407.07732","n_code_links":0,"syntology":null},{"paper":"/paper/a-fine-tuning-dataset-and-benchmark-for-large","slug":"a-fine-tuning-dataset-and-benchmark-for-large","title":"A Fine-tuning Dataset and Benchmark for Large Language Models for Protein Understanding","date":"2024-06-08","arxiv_id":"2406.05540","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tsynbio/proteinlmdataset"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"do-llms-recognize-me-when-i-is-not-me","title":"Do LLMs Recognize me, When I is not me: Assessment of LLMs Understanding of Turkish Indexical Pronouns in Indexical Shift Contexts","date":"2024-06-08","arxiv_id":"2406.05569","n_code_links":0,"syntology":null},{"paper":null,"slug":"selfdefend-llms-can-defend-themselves-against","title":"SelfDefend: LLMs Can Defend Themselves against Jailbreaking in a Practical Manner","date":"2024-06-08","arxiv_id":"2406.05498","n_code_links":0,"syntology":null},{"paper":null,"slug":"teaching-assistant-in-the-loop-improving","title":"Teaching-Assistant-in-the-Loop: Improving Knowledge Distillation from Imperfect Teacher Models in Low-Budget Scenarios","date":"2024-06-08","arxiv_id":"2406.05322","n_code_links":0,"syntology":null},{"paper":"/paper/toward-reliable-ad-hoc-scientific-information","slug":"toward-reliable-ad-hoc-scientific-information","title":"Toward Reliable Ad-hoc Scientific Information Extraction: A Case Study on Two Materials Datasets","date":"2024-06-08","arxiv_id":"2406.05348","n_code_links":1,"syntology":null},{"paper":null,"slug":"are-large-language-models-more-empathetic","title":"Are Large Language Models More Empathetic than Humans?","date":"2024-06-07","arxiv_id":"2406.05063","n_code_links":0,"syntology":null},{"paper":"/paper/diving-deep-into-the-motion-representation-of","slug":"diving-deep-into-the-motion-representation-of","title":"Diving Deep into the Motion Representation of Video-Text Models","date":"2024-06-07","arxiv_id":"2406.05075","n_code_links":1,"syntology":null},{"paper":"/paper/gamebench-evaluating-strategic-reasoning","slug":"gamebench-evaluating-strategic-reasoning","title":"GameBench: Evaluating Strategic Reasoning Abilities of LLM Agents","date":"2024-06-07","arxiv_id":"2406.06613","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Joshuaclymer/GameBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hints-in-browser-benchmarking-language-models","title":"Hints-In-Browser: Benchmarking Language Models for Programming Feedback Generation","date":"2024-06-07","arxiv_id":"2406.05053","n_code_links":0,"syntology":null},{"paper":"/paper/improving-logits-based-detector-without","slug":"improving-logits-based-detector-without","title":"DALD: Improving Logits-based Detector without Logits from Black-box LLMs","date":"2024-06-07","arxiv_id":"2406.05232","n_code_links":1,"syntology":null},{"paper":"/paper/llavaguard-vlm-based-safeguards-for-vision","slug":"llavaguard-vlm-based-safeguards-for-vision","title":"LlavaGuard: An Open VLM-based Framework for Safeguarding Vision Datasets and Models","date":"2024-06-07","arxiv_id":"2406.05113","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-research/llavaguard"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llms-are-not-intelligent-thinkers-introducing","slug":"llms-are-not-intelligent-thinkers-introducing","title":"LLMs Are Not Intelligent Thinkers: Introducing Mathematical Topic Tree Benchmark for Comprehensive Evaluation of LLMs","date":"2024-06-07","arxiv_id":"2406.05194","n_code_links":1,"syntology":null},{"paper":null,"slug":"low-resource-cross-lingual-summarization","title":"Low-Resource Cross-Lingual Summarization through Few-Shot Learning with Large Language Models","date":"2024-06-07","arxiv_id":"2406.04630","n_code_links":0,"syntology":null},{"paper":"/paper/mixture-of-agents-enhances-large-language","slug":"mixture-of-agents-enhances-large-language","title":"Mixture-of-Agents Enhances Large Language Model Capabilities","date":"2024-06-07","arxiv_id":"2406.04692","n_code_links":3,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"benchmark-data-contamination-of-large","title":"Benchmark Data Contamination of Large Language Models: A Survey","date":"2024-06-06","arxiv_id":"2406.04244","n_code_links":0,"syntology":null},{"paper":"/paper/characterizing-similarities-and-divergences","slug":"characterizing-similarities-and-divergences","title":"Characterizing Similarities and Divergences in Conversational Tones in Humans and LLMs by Sampling with People","date":"2024-06-06","arxiv_id":"2406.04278","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-latest-llms-for-leaderboard","title":"Exploring the Latest LLMs for Leaderboard Extraction","date":"2024-06-06","arxiv_id":"2406.04383","n_code_links":0,"syntology":null},{"paper":"/paper/generalization-enhanced-code-vulnerability","slug":"generalization-enhanced-code-vulnerability","title":"Generalization-Enhanced Code Vulnerability Detection via Multi-Task Instruction Fine-Tuning","date":"2024-06-06","arxiv_id":"2406.03718","n_code_links":1,"syntology":null},{"paper":null,"slug":"natural-plan-benchmarking-llms-on-natural","title":"NATURAL PLAN: Benchmarking LLMs on Natural Language Planning","date":"2024-06-06","arxiv_id":"2406.04520","n_code_links":0,"syntology":null},{"paper":null,"slug":"robocoder-robotic-learning-from-basic-skills","title":"RoboCoder: Robotic Learning from Basic Skills to General Tasks with Large Language Models","date":"2024-06-06","arxiv_id":"2406.03757","n_code_links":0,"syntology":null},{"paper":"/paper/scaling-and-evaluating-sparse-autoencoders","slug":"scaling-and-evaluating-sparse-autoencoders","title":"Scaling and evaluating sparse autoencoders","date":"2024-06-06","arxiv_id":"2406.04093","n_code_links":5,"syntology":{"ran":7,"of":10,"n_ran_checked":5,"n_instrument":2,"unverified":3,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["openai/sparse_autoencoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/tool-planner-dynamic-solution-tree-planning","slug":"tool-planner-dynamic-solution-tree-planning","title":"Tool-Planner: Task Planning with Clusters across Multiple Tools","date":"2024-06-06","arxiv_id":"2406.03807","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":1,"n_instrument":6,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["OceannTwT/Tool-Planner"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/ultramedical-building-specialized-generalists","slug":"ultramedical-building-specialized-generalists","title":"UltraMedical: Building Specialized Generalists in Biomedicine","date":"2024-06-06","arxiv_id":"2406.03949","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tsinghuac3i/ultramedical"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"analyzing-llm-behavior-in-dialogue","title":"Analyzing LLM Behavior in Dialogue Summarization: Unveiling Circumstantial Hallucination Trends","date":"2024-06-05","arxiv_id":"2406.03487","n_code_links":0,"syntology":null},{"paper":null,"slug":"biped-pedagogically-informed-tutoring-system","title":"BIPED: Pedagogically Informed Tutoring System for ESL Education","date":"2024-06-05","arxiv_id":"2406.03486","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-efficacy-of-large-language","title":"Evaluating the Efficacy of Large Language Models in Detecting Fake News: A Comparative Analysis","date":"2024-06-05","arxiv_id":"2406.06584","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-tarzan-to-tolkien-controlling-the","title":"From Tarzan to Tolkien: Controlling the Language Proficiency Level of LLMs for Content Generation","date":"2024-06-05","arxiv_id":"2406.03030","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-good-the-bad-and-the-hulk-like-gpt","title":"The Good, the Bad, and the Hulk-like GPT: Analyzing Emotional Decisions of Large Language Models in Cooperation and Bargaining Games","date":"2024-06-05","arxiv_id":"2406.03299","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-is-the-best-way-for-chatgpt-to-translate","title":"What is the Best Way for ChatGPT to Translate Poetry?","date":"2024-06-05","arxiv_id":"2406.03450","n_code_links":0,"syntology":null},{"paper":null,"slug":"dishonesty-in-helpful-and-harmless-alignment","title":"Dishonesty in Helpful and Harmless Alignment","date":"2024-06-04","arxiv_id":"2406.01931","n_code_links":0,"syntology":null},{"paper":null,"slug":"eliciting-the-priors-of-large-language-models","title":"Eliciting the Priors of Large Language Models using Iterated In-Context Learning","date":"2024-06-04","arxiv_id":"2406.01860","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-enabled-multi-agent","title":"Large Language Model-Enabled Multi-Agent Manufacturing Systems","date":"2024-06-04","arxiv_id":"2406.01893","n_code_links":0,"syntology":null},{"paper":"/paper/multiple-choice-questions-and-large-languages","slug":"multiple-choice-questions-and-large-languages","title":"Multiple Choice Questions and Large Languages Models: A Case Study with Fictional Medical Data","date":"2024-06-04","arxiv_id":"2406.02394","n_code_links":1,"syntology":null},{"paper":null,"slug":"badrag-identifying-vulnerabilities-in","title":"BadRAG: Identifying Vulnerabilities in Retrieval Augmented Generation of Large Language Models","date":"2024-06-03","arxiv_id":"2406.00083","n_code_links":0,"syntology":null},{"paper":"/paper/demystifying-platform-requirements-for","slug":"demystifying-platform-requirements-for","title":"Demystifying AI Platform Design for Distributed Inference of Next-Generation LLM models","date":"2024-06-03","arxiv_id":"2406.01698","n_code_links":1,"syntology":null},{"paper":"/paper/predicting-drug-gene-relations-via-analogy","slug":"predicting-drug-gene-relations-via-analogy","title":"Predicting Drug-Gene Relations via Analogy Tasks with Word Embeddings","date":"2024-06-03","arxiv_id":"2406.00984","n_code_links":1,"syntology":null},{"paper":null,"slug":"superhuman-performance-in-urology-board","title":"Superhuman performance in urology board questions by an explainable large language model enabled for context integration of the European Association of Urology guidelines: the UroBot study","date":"2024-06-03","arxiv_id":"2406.01428","n_code_links":0,"syntology":null},{"paper":null,"slug":"utilizing-large-language-models-for-1","title":"Utilizing Large Language Models for Automating Technical Customer Support","date":"2024-06-03","arxiv_id":"2406.01407","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-early-investigation-into-the-utility-of","title":"An Early Investigation into the Utility of Multimodal Large Language Models in Medical Imaging","date":"2024-06-02","arxiv_id":"2406.00667","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-mathematical-reasoning-of-large","slug":"evaluating-mathematical-reasoning-of-large","title":"Evaluating Mathematical Reasoning of Large Language Models: A Focus on Error Identification and Correction","date":"2024-06-02","arxiv_id":"2406.00755","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["littlecirc1e/eic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/presence-or-absence-are-unknown-word-usages","slug":"presence-or-absence-are-unknown-word-usages","title":"Presence or Absence: Are Unknown Word Usages in Dictionaries?","date":"2024-06-02","arxiv_id":"2406.00656","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-evaluation-benchmark-for-autoformalization","title":"An Evaluation Benchmark for Autoformalization in Lean4","date":"2024-06-01","arxiv_id":"2406.06555","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-metrics-evaluating-llms-effectiveness","title":"Beyond Metrics: Evaluating LLMs' Effectiveness in Culturally Nuanced, Low-Resource Real-World Scenarios","date":"2024-06-01","arxiv_id":"2406.00343","n_code_links":0,"syntology":null},{"paper":"/paper/phased-instruction-fine-tuning-for-large","slug":"phased-instruction-fine-tuning-for-large","title":"Phased Instruction Fine-Tuning for Large Language Models","date":"2024-06-01","arxiv_id":"2406.04371","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xubuvd/phasedsft"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"leveraging-large-language-models-for-entity","title":"Leveraging Large Language Models for Entity Matching","date":"2024-05-31","arxiv_id":"2405.20624","n_code_links":0,"syntology":null},{"paper":"/paper/query2cad-generating-cad-models-using-natural","slug":"query2cad-generating-cad-models-using-natural","title":"Query2CAD: Generating CAD models using natural language queries","date":"2024-05-31","arxiv_id":"2406.00144","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["akshay140601/query2cad"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/superlatives-in-context-explicit-and-implicit","slug":"superlatives-in-context-explicit-and-implicit","title":"Superlatives in Context: Modeling the Implicit Semantics of Superlatives","date":"2024-05-31","arxiv_id":"2405.20967","n_code_links":1,"syntology":null},{"paper":"/paper/video-mme-the-first-ever-comprehensive","slug":"video-mme-the-first-ever-comprehensive","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","date":"2024-05-31","arxiv_id":"2405.21075","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/an-automatic-question-usability-evaluation","slug":"an-automatic-question-usability-evaluation","title":"An Automatic Question Usability Evaluation Toolkit","date":"2024-05-30","arxiv_id":"2405.20529","n_code_links":1,"syntology":null},{"paper":"/paper/anah-analytical-annotation-of-hallucinations","slug":"anah-analytical-annotation-of-hallucinations","title":"ANAH: Analytical Annotation of Hallucinations in Large Language Models","date":"2024-05-30","arxiv_id":"2405.20315","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["open-compass/anah"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"autobreach-universal-and-adaptive","title":"AutoBreach: Universal and Adaptive Jailbreaking with Efficient Wordplay-Guided Optimization","date":"2024-05-30","arxiv_id":"2405.19668","n_code_links":0,"syntology":null},{"paper":"/paper/automated-generation-and-tagging-of-knowledge","slug":"automated-generation-and-tagging-of-knowledge","title":"Automated Generation and Tagging of Knowledge Components from Multiple-Choice Questions","date":"2024-05-30","arxiv_id":"2405.20526","n_code_links":1,"syntology":null},{"paper":null,"slug":"divide-and-conquer-meets-consensus-unleashing","title":"Divide-and-Conquer Meets Consensus: Unleashing the Power of Functions in Code Generation","date":"2024-05-30","arxiv_id":"2405.20092","n_code_links":0,"syntology":null},{"paper":"/paper/gnn-rag-graph-neural-retrieval-for-large","slug":"gnn-rag-graph-neural-retrieval-for-large","title":"GNN-RAG: Graph Neural Retrieval for Large Language Model Reasoning","date":"2024-05-30","arxiv_id":"2405.20139","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cmavro/gnn-rag"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/llamea-a-large-language-model-evolutionary","slug":"llamea-a-large-language-model-evolutionary","title":"LLaMEA: A Large Language Model Evolutionary Algorithm for Automatically Generating Metaheuristics","date":"2024-05-30","arxiv_id":"2405.20132","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nikivanstein/LLaMEA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"579b4e2779be7d90d4e700b22378491bb345a6c17d76259efffa5f8362588911","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}