{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/11","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":11,"pages_in_order":29,"rows_per_page":100,"rows":[1001,1100],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/10","next":"/method/gpt-4/papers/12","papers":[{"paper":null,"slug":"pron-vs-prompt-can-large-language-models","title":"Pron vs Prompt: Can Large Language Models already Challenge a World-Class Fiction Author at Creative Text Writing?","date":"2024-07-01","arxiv_id":"2407.01119","n_code_links":0,"syntology":null},{"paper":null,"slug":"roleplay-doh-enabling-domain-experts-to","title":"Roleplay-doh: Enabling Domain-Experts to Create LLM-simulated Patients via Eliciting and Adhering to Principles","date":"2024-07-01","arxiv_id":"2407.00870","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-bias-towards-medical","title":"Evaluation of Bias Towards Medical Professionals in Large Language Models","date":"2024-06-30","arxiv_id":"2407.12031","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-generated-natural-language-meets-scaling","title":"LLM-Generated Natural Language Meets Scaling Laws: New Explorations and Data Augmentation Methods","date":"2024-06-29","arxiv_id":"2407.00322","n_code_links":0,"syntology":null},{"paper":null,"slug":"too-late-to-train-too-early-to-use-a-study-on","title":"Too Late to Train, Too Early To Use? A Study on Necessity and Viability of Low-Resource Bengali LLMs","date":"2024-06-29","arxiv_id":"2407.00416","n_code_links":0,"syntology":null},{"paper":null,"slug":"urban-visual-appeal-according-to-chatgpt","title":"Urban Visual Appeal According to ChatGPT: Contrasting AI and Human Insights","date":"2024-06-29","arxiv_id":"2407.14268","n_code_links":0,"syntology":null},{"paper":"/paper/anomallmy-detecting-anomalous-tokens-in-black","slug":"anomallmy-detecting-anomalous-tokens-in-black","title":"AnomaLLMy -- Detecting anomalous tokens in black-box LLMs through low-confidence single-token predictions","date":"2024-06-28","arxiv_id":"2406.19840","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-gpt-4-help-detect-quit-vaping-intentions","title":"Can GPT-4 Help Detect Quit Vaping Intentions? An Exploration of Automatic Data Annotation Approach","date":"2024-06-28","arxiv_id":"2407.00167","n_code_links":0,"syntology":null},{"paper":null,"slug":"covert-malicious-finetuning-challenges-in","title":"Covert Malicious Finetuning: Challenges in Safeguarding LLM Adaptation","date":"2024-06-28","arxiv_id":"2406.20053","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-generate-high","slug":"can-large-language-models-generate-high","title":"Can Large Language Models Generate High-quality Patent Claims?","date":"2024-06-27","arxiv_id":"2406.19465","n_code_links":1,"syntology":null},{"paper":"/paper/sonnet-or-not-bot-poetry-evaluation-for-large","slug":"sonnet-or-not-bot-poetry-evaluation-for-large","title":"Sonnet or Not, Bot? Poetry Evaluation for Large Models and Datasets","date":"2024-06-27","arxiv_id":"2406.18906","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["maria-antoniak/poetry-eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-model-arena-for-cross-lingual-sentiment","title":"The Model Arena for Cross-lingual Sentiment Analysis: A Comparative Study in the Era of Large Language Models","date":"2024-06-27","arxiv_id":"2406.19358","n_code_links":0,"syntology":null},{"paper":"/paper/unigen-a-unified-framework-for-textual","slug":"unigen-a-unified-framework-for-textual","title":"UniGen: A Unified Framework for Textual Dataset Generation Using Large Language Models","date":"2024-06-27","arxiv_id":"2406.18966","n_code_links":1,"syntology":{"ran":8,"of":13,"n_ran_checked":8,"n_instrument":0,"unverified":5,"pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["howiehwong/unigen"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"adversarial-search-engine-optimization-for","title":"Adversarial Search Engine Optimization for Large Language Models","date":"2024-06-26","arxiv_id":"2406.18382","n_code_links":0,"syntology":null},{"paper":null,"slug":"apigen-automated-pipeline-for-generating","title":"APIGen: Automated Pipeline for Generating Verifiable and Diverse Function-Calling Datasets","date":"2024-06-26","arxiv_id":"2406.18518","n_code_links":0,"syntology":null},{"paper":"/paper/badge-badminton-report-generation-and","slug":"badge-badminton-report-generation-and","title":"BADGE: BADminton report Generation and Evaluation with LLM","date":"2024-06-26","arxiv_id":"2406.18116","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-entity-recognition-using-ensembles","title":"Improving Entity Recognition Using Ensembles of Deep Learning and Fine-tuned Large Language Models: A Case Study on Adverse Event Extraction from Multiple Sources","date":"2024-06-26","arxiv_id":"2406.18049","n_code_links":0,"syntology":null},{"paper":"/paper/jailbreaking-llms-with-arabic-transliteration","slug":"jailbreaking-llms-with-arabic-transliteration","title":"Jailbreaking LLMs with Arabic Transliteration and Arabizi","date":"2024-06-26","arxiv_id":"2406.18725","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["securedl/arabic_jailbreak"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"octo-planner-on-device-language-model-for","title":"Octo-planner: On-device Language Model for Planner-Action Agents","date":"2024-06-26","arxiv_id":"2406.18082","n_code_links":0,"syntology":null},{"paper":null,"slug":"re-ranking-step-by-step-investigating-pre","title":"Re-Ranking Step by Step: Investigating Pre-Filtering for Re-Ranking with Large Language Models","date":"2024-06-26","arxiv_id":"2406.18740","n_code_links":0,"syntology":null},{"paper":"/paper/step-dpo-step-wise-preference-optimization","slug":"step-dpo-step-wise-preference-optimization","title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","date":"2024-06-26","arxiv_id":"2406.18629","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":10,"n_instrument":1,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dvlab-research/step-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/themis-towards-flexible-and-interpretable-nlg","slug":"themis-towards-flexible-and-interpretable-nlg","title":"Themis: A Reference-free NLG Evaluation Language Model with Flexibility and Interpretability","date":"2024-06-26","arxiv_id":"2406.18365","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":3,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["PKU-ONELab/Themis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/wildguard-open-one-stop-moderation-tools-for","slug":"wildguard-open-one-stop-moderation-tools-for","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","date":"2024-06-26","arxiv_id":"2406.18495","n_code_links":4,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["allenai/wildguard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"accelerating-clinical-evidence-synthesis-with","title":"Accelerating Clinical Evidence Synthesis with Large Language Models","date":"2024-06-25","arxiv_id":"2406.17755","n_code_links":0,"syntology":null},{"paper":"/paper/ares-alternating-reinforcement-learning-and","slug":"ares-alternating-reinforcement-learning-and","title":"ARES: Alternating Reinforcement Learning and Supervised Fine-Tuning for Enhanced Multi-Modal Chain-of-Thought Reasoning Through Diverse AI Feedback","date":"2024-06-25","arxiv_id":"2407.00087","n_code_links":1,"syntology":null},{"paper":null,"slug":"autonomous-prompt-engineering-in-large","title":"Autonomous Prompt Engineering in Large Language Models","date":"2024-06-25","arxiv_id":"2407.11000","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-distillation-in-automated","title":"Knowledge Distillation in Automated Annotation: Supervised Text Classification with LLM-Generated Training Labels","date":"2024-06-25","arxiv_id":"2406.17633","n_code_links":0,"syntology":null},{"paper":null,"slug":"longins-a-challenging-long-context","title":"LongIns: A Challenging Long-context Instruction-based Exam for LLMs","date":"2024-06-25","arxiv_id":"2406.17588","n_code_links":0,"syntology":null},{"paper":null,"slug":"anomaly-detection-of-tabular-data-using-llms","title":"Anomaly Detection of Tabular Data Using LLMs","date":"2024-06-24","arxiv_id":"2406.16308","n_code_links":0,"syntology":null},{"paper":null,"slug":"classification-of-geological-borehole","title":"Classification of Geological Borehole Descriptions Using a Domain Adapted Large Language Model","date":"2024-06-24","arxiv_id":"2407.10991","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-factual-entailment-with-nli-a-news","title":"Exploring Factual Entailment with NLI: A News Media Study","date":"2024-06-24","arxiv_id":"2406.16842","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-in-student-assessment","title":"Large Language Models in Student Assessment: Comparing ChatGPT and Human Graders","date":"2024-06-24","arxiv_id":"2406.16510","n_code_links":0,"syntology":null},{"paper":"/paper/multi-logieval-towards-evaluating-multi-step","slug":"multi-logieval-towards-evaluating-multi-step","title":"Multi-LogiEval: Towards Evaluating Multi-Step Logical Reasoning Ability of Large Language Models","date":"2024-06-24","arxiv_id":"2406.17169","n_code_links":1,"syntology":null},{"paper":null,"slug":"plagbench-exploring-the-duality-of-large","title":"PlagBench: Exploring the Duality of Large Language Models in Plagiarism Generation and Detection","date":"2024-06-24","arxiv_id":"2406.16288","n_code_links":0,"syntology":null},{"paper":null,"slug":"uno-arena-for-evaluating-sequential-decision","title":"UNO Arena for Evaluating Sequential Decision-Making Capability of Large Language Models","date":"2024-06-24","arxiv_id":"2406.16382","n_code_links":0,"syntology":null},{"paper":null,"slug":"usdc-a-dataset-of-underline-u-ser-underline-s","title":"USDC: A Dataset of $\\underline{U}$ser $\\underline{S}$tance and $\\underline{D}$ogmatism in Long $\\underline{C}$onversations","date":"2024-06-24","arxiv_id":"2406.16833","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-commentary-strategies-for-imperfect","slug":"enhancing-commentary-strategies-for-imperfect","title":"Enhancing Commentary Strategies for Imperfect Information Card Games: A Study of Large Language Models in Guandan Commentary","date":"2024-06-23","arxiv_id":"2406.17807","n_code_links":1,"syntology":null},{"paper":null,"slug":"grapheval2000-benchmarking-and-improving","title":"GraphEval2000: Benchmarking and Improving Large Language Models on Graph Datasets","date":"2024-06-23","arxiv_id":"2406.16176","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-generate-visualizations-with","title":"Can LLMs Generate Visualizations with Dataless Prompts?","date":"2024-06-22","arxiv_id":"2406.17805","n_code_links":0,"syntology":null},{"paper":"/paper/ladder-a-model-agnostic-framework-boosting","slug":"ladder-a-model-agnostic-framework-boosting","title":"Ladder: A Model-Agnostic Framework Boosting LLM-based Machine Translation to the Next Level","date":"2024-06-22","arxiv_id":"2406.15741","n_code_links":3,"syntology":{"ran":2,"of":8,"n_ran_checked":0,"n_instrument":2,"unverified":6,"pointer_only":8,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["fzp0424/ladder","fzp0424/mt-ladder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-gpt-based-code-review-system-for","title":"A GPT-based Code Review System for Programming Language Learning","date":"2024-06-21","arxiv_id":"2407.04722","n_code_links":0,"syntology":null},{"paper":"/paper/a-smart-mnemonic-sounds-like-glue-tonic","slug":"a-smart-mnemonic-sounds-like-glue-tonic","title":"A SMART Mnemonic Sounds like \"Glue Tonic\": Mixing LLMs with Student Feedback to Make Mnemonic Learning Stick","date":"2024-06-21","arxiv_id":"2406.15352","n_code_links":1,"syntology":null},{"paper":null,"slug":"data-efficient-evaluation-of-large-language","title":"Data Efficient Evaluation of Large Language Models and Text-to-Image Models via Adaptive Sampling","date":"2024-06-21","arxiv_id":"2406.15527","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-continual-pre-training-by","title":"Efficient Continual Pre-training by Mitigating the Stability Gap","date":"2024-06-21","arxiv_id":"2406.14833","n_code_links":0,"syntology":null},{"paper":"/paper/esc-eval-evaluating-emotion-support","slug":"esc-eval-evaluating-emotion-support","title":"ESC-Eval: Evaluating Emotion Support Conversations in Large Language Models","date":"2024-06-21","arxiv_id":"2406.14952","n_code_links":3,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["aiflames/esc-eval","haidequanbu/esc-eval","smartflowai/emollm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"how-effective-is-gpt-4-turbo-in-generating","title":"How Effective is GPT-4 Turbo in Generating School-Level Questions from Textbooks Based on Bloom's Revised Taxonomy?","date":"2024-06-21","arxiv_id":"2406.15211","n_code_links":0,"syntology":null},{"paper":"/paper/internlm-law-an-open-source-chinese-legal","slug":"internlm-law-an-open-source-chinese-legal","title":"InternLM-Law: An Open Source Chinese Legal Large Language Model","date":"2024-06-21","arxiv_id":"2406.14887","n_code_links":1,"syntology":null},{"paper":"/paper/tinystyler-efficient-few-shot-text-style","slug":"tinystyler-efficient-few-shot-text-style","title":"TinyStyler: Efficient Few-Shot Text Style Transfer with Authorship Embeddings","date":"2024-06-21","arxiv_id":"2406.15586","n_code_links":1,"syntology":null},{"paper":"/paper/v-recs-a-low-cost-llm4vis-recommender-with","slug":"v-recs-a-low-cost-llm4vis-recommender-with","title":"V-RECS, a Low-Cost LLM4VIS Recommender with Explanations, Captioning and Suggestions","date":"2024-06-21","arxiv_id":"2406.15259","n_code_links":1,"syntology":null},{"paper":null,"slug":"2406-15508","title":"What Teaches Robots to Walk, Teaches Them to Trade too -- Regime Adaptive Execution using Informed Data and LLMs","date":"2024-06-20","arxiv_id":"2406.15508","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-large-language-model-outperforms-other","title":"A Large Language Model Outperforms Other Computational Approaches to the High-Throughput Phenotyping of Physician Notes","date":"2024-06-20","arxiv_id":"2406.14757","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgpt-as-research-scientist-probing-gpt-s","title":"ChatGPT as Research Scientist: Probing GPT's Capabilities as a Research Librarian, Research Ethicist, Data Generator and Data Predictor","date":"2024-06-20","arxiv_id":"2406.14765","n_code_links":0,"syntology":null},{"paper":null,"slug":"cryptogpt-a-7b-model-rivaling-gpt-4-in-the","title":"CryptoGPT: a 7B model rivaling GPT-4 in the task of analyzing and classifying real-time financial news","date":"2024-06-20","arxiv_id":"2406.14039","n_code_links":0,"syntology":null},{"paper":"/paper/diras-efficient-llm-assisted-annotation-of","slug":"diras-efficient-llm-assisted-annotation-of","title":"DIRAS: Efficient LLM Annotation of Document Relevance in Retrieval Augmented Generation","date":"2024-06-20","arxiv_id":"2406.14162","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-the-llm-based-robot-manipulation","title":"Enhancing the LLM-Based Robot Manipulation Through Human-Robot Collaboration","date":"2024-06-20","arxiv_id":"2406.14097","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-for-enhancing-active-learning","title":"Generative AI for Enhancing Active Learning in Education: A Comparative Study of GPT-3.5 and GPT-4 in Crafting Customized Test Questions","date":"2024-06-20","arxiv_id":"2406.13903","n_code_links":0,"syntology":null},{"paper":null,"slug":"identifying-user-goals-from-ui-trajectories","title":"Identifying User Goals from UI Trajectories","date":"2024-06-20","arxiv_id":"2406.14314","n_code_links":0,"syntology":null},{"paper":"/paper/mmbench-video-a-long-form-multi-shot","slug":"mmbench-video-a-long-form-multi-shot","title":"MMBench-Video: A Long-Form Multi-Shot Benchmark for Holistic Video Understanding","date":"2024-06-20","arxiv_id":"2406.14515","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":1,"n_instrument":4,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["open-compass/vlmevalkit"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mr-ben-a-comprehensive-meta-reasoning","title":"MR-Ben: A Meta-Reasoning Benchmark for Evaluating System-2 Thinking in LLMs","date":"2024-06-20","arxiv_id":"2406.13975","n_code_links":0,"syntology":null},{"paper":"/paper/sorry-bench-systematically-evaluating-large","slug":"sorry-bench-systematically-evaluating-large","title":"SORRY-Bench: Systematically Evaluating Large Language Model Safety Refusal Behaviors","date":"2024-06-20","arxiv_id":"2406.14598","n_code_links":1,"syntology":null},{"paper":null,"slug":"spl-a-socratic-playground-for-learning","title":"SPL: A Socratic Playground for Learning Powered by Large Language Model","date":"2024-06-20","arxiv_id":"2406.13919","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-use-of-multimodal-large-language-models","title":"The Use of Multimodal Large Language Models to Detect Objects from Thermal Images: Transportation Applications","date":"2024-06-20","arxiv_id":"2406.13898","n_code_links":0,"syntology":null},{"paper":null,"slug":"ttqa-rs-a-break-down-prompting-approach-for","title":"TTQA-RS- A break-down prompting approach for Multi-hop Table-Text Question Answering with Reasoning and Summarization","date":"2024-06-20","arxiv_id":"2406.14732","n_code_links":0,"syntology":null},{"paper":null,"slug":"unmasking-database-vulnerabilities-zero","title":"Unmasking Database Vulnerabilities: Zero-Knowledge Schema Inference Attacks in Text-to-SQL Systems","date":"2024-06-20","arxiv_id":"2406.14545","n_code_links":0,"syntology":null},{"paper":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","slug":"alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","arxiv_id":"2406.13807","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alanaai/evud"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/do-multimodal-foundation-models-understand","slug":"do-multimodal-foundation-models-understand","title":"WONDERBREAD: A Benchmark for Evaluating Multimodal Foundation Models on Business Process Management Tasks","date":"2024-06-19","arxiv_id":"2406.13264","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hazyresearch/wonderbread"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"is-gpt-4-conscious","title":"Is GPT-4 conscious?","date":"2024-06-19","arxiv_id":"2407.09517","n_code_links":0,"syntology":null},{"paper":"/paper/morehopqa-more-than-multi-hop-reasoning","slug":"morehopqa-more-than-multi-hop-reasoning","title":"MoreHopQA: More Than Multi-hop Reasoning","date":"2024-06-19","arxiv_id":"2406.13397","n_code_links":1,"syntology":null},{"paper":"/paper/an-investigation-of-neuron-activation-as-a","slug":"an-investigation-of-neuron-activation-as-a","title":"An Investigation of Neuron Activation as a Unified Lens to Explain Chain-of-Thought Eliciting Arithmetic Reasoning of LLMs","date":"2024-06-18","arxiv_id":"2406.12288","n_code_links":2,"syntology":{"ran":34,"of":36,"n_ran_checked":32,"n_instrument":2,"unverified":2,"pointer_only":12,"phrase":"34 ran (of which 0 constructed an object rather than computing a result; 32 with no instrument failure: 0 honoured, 3 violated, 29 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dakingrai/neuron-analysis-cot-arithmetic-reasoning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"assessing-ai-vs-human-authored-spear-phishing","title":"Assessing AI vs Human-Authored Spear Phishing SMS Attacks: An Empirical Study","date":"2024-06-18","arxiv_id":"2406.13049","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-always-solve-easy","slug":"can-large-language-models-always-solve-easy","title":"Can Large Language Models Always Solve Easy Problems if They Can Solve Harder Ones?","date":"2024-06-18","arxiv_id":"2406.12809","n_code_links":1,"syntology":{"ran":11,"of":14,"n_ran_checked":11,"n_instrument":0,"unverified":3,"pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["QwenLM/ConsisEval"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatglm-a-family-of-large-language-models","slug":"chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","n_code_links":7,"syntology":{"ran":21,"of":29,"n_ran_checked":20,"n_instrument":1,"unverified":8,"pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["thudm/chatglm-6b"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/dart-math-difficulty-aware-rejection-tuning-1","slug":"dart-math-difficulty-aware-rejection-tuning-1","title":"DART-Math: Difficulty-Aware Rejection Tuning for Mathematical Problem-Solving","date":"2024-06-18","arxiv_id":"2407.13690","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hkust-nlp/dart-math"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-artificial-intelligence-guided","title":"Generative Artificial Intelligence-Guided User Studies: An Application for Air Taxi Services","date":"2024-06-18","arxiv_id":"2406.12296","n_code_links":0,"syntology":null},{"paper":"/paper/interpretable-preferences-via-multi-objective","slug":"interpretable-preferences-via-multi-objective","title":"Interpretable Preferences via Multi-Objective Reward Modeling and Mixture-of-Experts","date":"2024-06-18","arxiv_id":"2406.12845","n_code_links":2,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["RLHFlow/RLHF-Reward-Modeling"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/judging-the-judges-evaluating-alignment-and","slug":"judging-the-judges-evaluating-alignment-and","title":"Judging the Judges: Evaluating Alignment and Vulnerabilities in LLMs-as-Judges","date":"2024-06-18","arxiv_id":"2406.12624","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["UMass-Meta-LLM-Eval/llm_eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/measuring-psychological-depth-in-language","slug":"measuring-psychological-depth-in-language","title":"Measuring Psychological Depth in Language Models","date":"2024-06-18","arxiv_id":"2406.12680","n_code_links":1,"syntology":null},{"paper":"/paper/ubench-benchmarking-uncertainty-in-large","slug":"ubench-benchmarking-uncertainty-in-large","title":"UBENCH: Benchmarking Uncertainty in Large Language Models with Multiple Choice Questions","date":"2024-06-18","arxiv_id":"2406.12784","n_code_links":1,"syntology":null},{"paper":null,"slug":"vernacular-i-barely-know-her-challenges-with","title":"Vernacular? I Barely Know Her: Challenges with Style Control and Stereotyping","date":"2024-06-18","arxiv_id":"2406.12679","n_code_links":0,"syntology":null},{"paper":"/paper/a-two-dimensional-zero-shot-dialogue-state","slug":"a-two-dimensional-zero-shot-dialogue-state","title":"A Two-dimensional Zero-shot Dialogue State Tracking Evaluation Method using GPT-4","date":"2024-06-17","arxiv_id":"2406.11651","n_code_links":1,"syntology":null},{"paper":"/paper/are-large-language-models-true-healthcare","slug":"are-large-language-models-true-healthcare","title":"Are Large Language Models True Healthcare Jacks-of-All-Trades? Benchmarking Across Health Professions Beyond Physician Exams","date":"2024-06-17","arxiv_id":"2406.11328","n_code_links":1,"syntology":null},{"paper":"/paper/beyond-boundaries-learning-a-universal-entity","slug":"beyond-boundaries-learning-a-universal-entity","title":"Beyond Boundaries: Learning a Universal Entity Taxonomy across Datasets and Languages for Open Named Entity Recognition","date":"2024-06-17","arxiv_id":"2406.11192","n_code_links":1,"syntology":null},{"paper":null,"slug":"cultural-conditioning-or-placebo-on-the","title":"Cultural Conditioning or Placebo? On the Effectiveness of Socio-Demographic Prompting","date":"2024-06-17","arxiv_id":"2406.11661","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-the-narratives-analyzing-personal","title":"Decoding the Narratives: Analyzing Personal Drug Experiences Shared on Reddit","date":"2024-06-17","arxiv_id":"2406.12117","n_code_links":0,"syntology":null},{"paper":null,"slug":"enabling-robots-to-follow-abstract","title":"Enabling robots to follow abstract instructions and complete complex dynamic tasks","date":"2024-06-17","arxiv_id":"2406.11231","n_code_links":0,"syntology":null},{"paper":"/paper/fintruthqa-a-benchmark-dataset-for-evaluating","slug":"fintruthqa-a-benchmark-dataset-for-evaluating","title":"FinTruthQA: A Benchmark Dataset for Evaluating the Quality of Financial Information Disclosure","date":"2024-06-17","arxiv_id":"2406.12009","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bethxx99/FinTruthQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/geogpt4v-towards-geometric-multi-modal-large","slug":"geogpt4v-towards-geometric-multi-modal-large","title":"GeoGPT4V: Towards Geometric Multi-modal Large Language Models with Geometric Image Generation","date":"2024-06-17","arxiv_id":"2406.11503","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lanyu0303/geogpt4v_project"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/global-data-constraints-ethical-and","slug":"global-data-constraints-ethical-and","title":"Problematic Tokens: Tokenizer Bias in Large Language Models","date":"2024-06-17","arxiv_id":"2406.11214","n_code_links":1,"syntology":null},{"paper":null,"slug":"iterative-length-regularized-direct","title":"Iterative Length-Regularized Direct Preference Optimization: A Case Study on Improving 7B Language Models to GPT-4 Level","date":"2024-06-17","arxiv_id":"2406.11817","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-and-knowledge-graphs-1","title":"Large Language Models and Knowledge Graphs for Astronomical Entity Disambiguation","date":"2024-06-17","arxiv_id":"2406.11400","n_code_links":0,"syntology":null},{"paper":null,"slug":"metagpt-merging-large-language-models-using","title":"MetaGPT: Merging Large Language Models Using Model Exclusive Task Arithmetic","date":"2024-06-17","arxiv_id":"2406.11385","n_code_links":0,"syntology":null},{"paper":"/paper/satyrn-a-platform-for-analytics-augmented","slug":"satyrn-a-platform-for-analytics-augmented","title":"Satyrn: A Platform for Analytics Augmented Generation","date":"2024-06-17","arxiv_id":"2406.12069","n_code_links":1,"syntology":null},{"paper":null,"slug":"should-ai-optimize-your-code-a-comparative","title":"Should AI Optimize Your Code? A Comparative Study of Classical Optimizing Compilers Versus Current Large Language Models","date":"2024-06-17","arxiv_id":"2406.12146","n_code_links":0,"syntology":null},{"paper":"/paper/small-agent-can-also-rock-empowering-small","slug":"small-agent-can-also-rock-empowering-small","title":"Small Agent Can Also Rock! Empowering Small Language Models as Hallucination Detector","date":"2024-06-17","arxiv_id":"2406.11277","n_code_links":1,"syntology":null},{"paper":"/paper/unveiling-and-mitigating-bias-in-mental","slug":"unveiling-and-mitigating-bias-in-mental","title":"Unveiling and Mitigating Bias in Mental Health Analysis with Large Language Models","date":"2024-06-17","arxiv_id":"2406.12033","n_code_links":1,"syntology":null},{"paper":"/paper/can-llms-understand-the-implication-of","slug":"can-llms-understand-the-implication-of","title":"Can LLMs Understand the Implication of Emphasized Sentences in Dialogue?","date":"2024-06-16","arxiv_id":"2406.11065","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-opinions-at-scale-incremental","title":"Distilling Opinions at Scale: Incremental Opinion Summarization using XL-OPSUMM","date":"2024-06-16","arxiv_id":"2406.10886","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-supermarket-robot-interaction-a","title":"Enhancing Supermarket Robot Interaction: A Multi-Level LLM Conversational Interface for Handling Diverse Customer Intents","date":"2024-06-16","arxiv_id":"2406.11047","n_code_links":0,"syntology":null},{"paper":null,"slug":"exposing-the-achilles-heel-evaluating-llms","title":"Exposing the Achilles' Heel: Evaluating LLMs Ability to Handle Mistakes in Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10834","n_code_links":0,"syntology":null},{"paper":"/paper/generating-tables-from-the-parametric","slug":"generating-tables-from-the-parametric","title":"Generating Tables from the Parametric Knowledge of Language Models","date":"2024-06-16","arxiv_id":"2406.10922","n_code_links":1,"syntology":null}],"record_sha256":"92296dc1885eea9db63dffa78cdab28c681d32dc86b9f35009617ca2b7061bbf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}