{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/5","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":5,"pages_in_order":20,"rows_per_page":100,"rows":[401,500],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/4","next":"/method/gpt-3/papers/6","papers":[{"paper":null,"slug":"evaluating-the-effectiveness-of-the","title":"Evaluating the Effectiveness of the Foundational Models for Q&A Classification in Mental Health care","date":"2024-06-23","arxiv_id":"2406.15966","n_code_links":0,"syntology":null},{"paper":null,"slug":"grapheval2000-benchmarking-and-improving","title":"GraphEval2000: Benchmarking and Improving Large Language Models on Graph Datasets","date":"2024-06-23","arxiv_id":"2406.16176","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-generate-visualizations-with","title":"Can LLMs Generate Visualizations with Dataless Prompts?","date":"2024-06-22","arxiv_id":"2406.17805","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgpt-as-research-scientist-probing-gpt-s","title":"ChatGPT as Research Scientist: Probing GPT's Capabilities as a Research Librarian, Research Ethicist, Data Generator and Data Predictor","date":"2024-06-20","arxiv_id":"2406.14765","n_code_links":0,"syntology":null},{"paper":null,"slug":"cryptogpt-a-7b-model-rivaling-gpt-4-in-the","title":"CryptoGPT: a 7B model rivaling GPT-4 in the task of analyzing and classifying real-time financial news","date":"2024-06-20","arxiv_id":"2406.14039","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-implicit-bias-in-large-language","slug":"evaluating-implicit-bias-in-large-language","title":"Evaluating Implicit Bias in Large Language Models by Attacking From a Psychometric Perspective","date":"2024-06-20","arxiv_id":"2406.14023","n_code_links":1,"syntology":null},{"paper":null,"slug":"generative-ai-for-enhancing-active-learning","title":"Generative AI for Enhancing Active Learning in Education: A Comparative Study of GPT-3.5 and GPT-4 in Crafting Customized Test Questions","date":"2024-06-20","arxiv_id":"2406.13903","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-plan-for-retrieval-augmented","slug":"learning-to-plan-for-retrieval-augmented","title":"Learning to Plan for Retrieval-Augmented Large Language Models from Knowledge Graphs","date":"2024-06-20","arxiv_id":"2406.14282","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zjukg/lpkg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llasa-large-multimodal-agent-for-human","slug":"llasa-large-multimodal-agent-for-human","title":"LLaSA: A Multimodal LLM for Human Activity Analysis Through Wearable and Smartphone Sensors","date":"2024-06-20","arxiv_id":"2406.14498","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":3,"n_instrument":3,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bashlab/llasa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/prism-a-framework-for-decoupling-and","slug":"prism-a-framework-for-decoupling-and","title":"Prism: A Framework for Decoupling and Assessing the Capabilities of VLMs","date":"2024-06-20","arxiv_id":"2406.14544","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["sparksjoe/prism"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"scidmt-a-large-scale-corpus-for-detecting","title":"SciDMT: A Large-Scale Corpus for Detecting Scientific Mentions","date":"2024-06-20","arxiv_id":"2406.14756","n_code_links":0,"syntology":null},{"paper":"/paper/part-aware-unified-representation-of-language-1","slug":"part-aware-unified-representation-of-language-1","title":"Part-aware Unified Representation of Language and Skeleton for Zero-shot Action Recognition","date":"2024-06-19","arxiv_id":"2406.13327","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["azzh1/purls"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generating-educational-materials-with","title":"Generating Educational Materials with Different Levels of Readability using LLMs","date":"2024-06-18","arxiv_id":"2406.12787","n_code_links":0,"syntology":null},{"paper":"/paper/towards-a-client-centered-assessment-of-llm","slug":"towards-a-client-centered-assessment-of-llm","title":"Towards a Client-Centered Assessment of LLM Therapists by Client Simulation","date":"2024-06-18","arxiv_id":"2406.12266","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wangjs9/clientcast"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vernacular-i-barely-know-her-challenges-with","title":"Vernacular? I Barely Know Her: Challenges with Style Control and Stereotyping","date":"2024-06-18","arxiv_id":"2406.12679","n_code_links":0,"syntology":null},{"paper":null,"slug":"you-gotta-be-a-doctor-lin-an-investigation-of","title":"\"You Gotta be a Doctor, Lin\": An Investigation of Name-Based Bias of Large Language Models in Employment Recommendations","date":"2024-06-18","arxiv_id":"2406.12232","n_code_links":0,"syntology":null},{"paper":"/paper/building-another-spanish-dictionary-this-time","slug":"building-another-spanish-dictionary-this-time","title":"Building another Spanish dictionary, this time with GPT-4","date":"2024-06-17","arxiv_id":"2406.11218","n_code_links":1,"syntology":null},{"paper":null,"slug":"cultural-conditioning-or-placebo-on-the","title":"Cultural Conditioning or Placebo? On the Effectiveness of Socio-Demographic Prompting","date":"2024-06-17","arxiv_id":"2406.11661","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-biomedical-knowledge-retrieval","title":"SeRTS: Self-Rewarding Tree Search for Biomedical Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11258","n_code_links":0,"syntology":null},{"paper":"/paper/enhancing-text-classification-through-llm","slug":"enhancing-text-classification-through-llm","title":"Enhancing Text Classification through LLM-Driven Active Learning and Human Annotation","date":"2024-06-17","arxiv_id":"2406.12114","n_code_links":1,"syntology":null},{"paper":null,"slug":"estimating-the-increase-in-emissions-caused","title":"Estimating the Increase in Emissions caused by AI-augmented Search","date":"2024-06-17","arxiv_id":"2407.16894","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-open-language-models-across-task","slug":"evaluating-open-language-models-across-task","title":"Are Small Language Models Ready to Compete with Large Language Models for Practical Applications?","date":"2024-06-17","arxiv_id":"2406.11402","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-safety-utility-trade-offs-in","title":"Exploring Safety-Utility Trade-Offs in Personalized Language Models","date":"2024-06-17","arxiv_id":"2406.11107","n_code_links":0,"syntology":null},{"paper":"/paper/investigating-annotator-bias-in-large","slug":"investigating-annotator-bias-in-large","title":"Investigating Annotator Bias in Large Language Models for Hate Speech Detection","date":"2024-06-17","arxiv_id":"2406.11109","n_code_links":3,"syntology":null},{"paper":null,"slug":"jobfair-a-framework-for-benchmarking-gender","title":"JobFair: A Framework for Benchmarking Gender Hiring Bias in Large Language Models","date":"2024-06-17","arxiv_id":"2406.15484","n_code_links":0,"syntology":null},{"paper":"/paper/welldunn-on-the-robustness-and-explainability","slug":"welldunn-on-the-robustness-and-explainability","title":"WellDunn: On the Robustness and Explainability of Language Models and Large Language Models in Identifying Wellness Dimensions","date":"2024-06-17","arxiv_id":"2406.12058","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vedantpalit/WellDunn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exposing-the-achilles-heel-evaluating-llms","title":"Exposing the Achilles' Heel: Evaluating LLMs Ability to Handle Mistakes in Mathematical Reasoning","date":"2024-06-16","arxiv_id":"2406.10834","n_code_links":0,"syntology":null},{"paper":"/paper/generating-tables-from-the-parametric","slug":"generating-tables-from-the-parametric","title":"Generating Tables from the Parametric Knowledge of Language Models","date":"2024-06-16","arxiv_id":"2406.10922","n_code_links":1,"syntology":null},{"paper":null,"slug":"grading-massive-open-online-courses-using","title":"Grading Massive Open Online Courses Using Large Language Models","date":"2024-06-16","arxiv_id":"2406.11102","n_code_links":0,"syntology":null},{"paper":"/paper/kgpa-robustness-evaluation-for-large-language","slug":"kgpa-robustness-evaluation-for-large-language","title":"KGPA: Robustness Evaluation for Large Language Models via Cross-Domain Knowledge Graphs","date":"2024-06-16","arxiv_id":"2406.10802","n_code_links":1,"syntology":null},{"paper":"/paper/beyond-raw-videos-understanding-edited-videos","slug":"beyond-raw-videos-understanding-edited-videos","title":"Beyond Raw Videos: Understanding Edited Videos with Large Multimodal Model","date":"2024-06-15","arxiv_id":"2406.10484","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-the-correlation-between-human-and","title":"Exploring the Correlation between Human and Machine Evaluation of Simultaneous Speech Translation","date":"2024-06-14","arxiv_id":"2406.10091","n_code_links":0,"syntology":null},{"paper":null,"slug":"linguistic-bias-in-chatgpt-language-models","title":"Linguistic Bias in ChatGPT: Language Models Reinforce Dialect Discrimination","date":"2024-06-13","arxiv_id":"2406.08818","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuned-small-llms-still-significantly","slug":"fine-tuned-small-llms-still-significantly","title":"Fine-Tuned 'Small' LLMs (Still) Significantly Outperform Zero-Shot Generative AI Models in Text Classification","date":"2024-06-12","arxiv_id":"2406.08660","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mnbucher/text-cls-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"how-well-it-works-benchmarking-performance-of","title":"How well it works: Benchmarking performance of GPT models on medical natural language processing tasks","date":"2024-06-12","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/indirectrequests-making-task-oriented","slug":"indirectrequests-making-task-oriented","title":"Making Task-Oriented Dialogue Datasets More Natural by Synthetically Generating Indirect User Requests","date":"2024-06-12","arxiv_id":"2406.07794","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-words-on-large-language-models","title":"Beyond Words: On Large Language Models Actionability in Mission-Critical Risk Analysis","date":"2024-06-11","arxiv_id":"2406.10273","n_code_links":0,"syntology":null},{"paper":null,"slug":"bilingual-sexism-classification-fine-tuned","title":"Bilingual Sexism Classification: Fine-Tuned XLM-RoBERTa and GPT-3.5 Few-Shot Learning","date":"2024-06-11","arxiv_id":"2406.07287","n_code_links":0,"syntology":null},{"paper":null,"slug":"flextron-many-in-one-flexible-large-language","title":"Flextron: Many-in-One Flexible Large Language Model","date":"2024-06-11","arxiv_id":"2406.10260","n_code_links":0,"syntology":null},{"paper":"/paper/multi-objective-reinforcement-learning-from","slug":"multi-objective-reinforcement-learning-from","title":"Multi-objective Reinforcement learning from AI Feedback","date":"2024-06-11","arxiv_id":"2406.07295","n_code_links":1,"syntology":null},{"paper":"/paper/agb-de-a-corpus-for-the-automated-legal","slug":"agb-de-a-corpus-for-the-automated-legal","title":"AGB-DE: A Corpus for the Automated Legal Assessment of Clauses in German Consumer Contracts","date":"2024-06-10","arxiv_id":"2406.06809","n_code_links":1,"syntology":null},{"paper":"/paper/in-context-learning-and-fine-tuning-gpt-for","slug":"in-context-learning-and-fine-tuning-gpt-for","title":"In-Context Learning and Fine-Tuning GPT for Argument Mining","date":"2024-06-10","arxiv_id":"2406.06699","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-llms-recognize-me-when-i-is-not-me","title":"Do LLMs Recognize me, When I is not me: Assessment of LLMs Understanding of Turkish Indexical Pronouns in Indexical Shift Contexts","date":"2024-06-08","arxiv_id":"2406.05569","n_code_links":0,"syntology":null},{"paper":null,"slug":"selfdefend-llms-can-defend-themselves-against","title":"SelfDefend: LLMs Can Defend Themselves against Jailbreaking in a Practical Manner","date":"2024-06-08","arxiv_id":"2406.05498","n_code_links":0,"syntology":null},{"paper":"/paper/bamo-at-semeval-2024-task-9-brainteaser-a","slug":"bamo-at-semeval-2024-task-9-brainteaser-a","title":"BAMO at SemEval-2024 Task 9: BRAINTEASER: A Novel Task Defying Common Sense","date":"2024-06-07","arxiv_id":"2406.04947","n_code_links":1,"syntology":null},{"paper":"/paper/berts-are-generative-in-context-learners","slug":"berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","arxiv_id":"2406.04823","n_code_links":1,"syntology":{"ran":13,"of":26,"n_ran_checked":12,"n_instrument":1,"unverified":13,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 13 unverified","official":{"repos":["ltgoslo/bert-in-context"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":13,"ran_from_kinds":["official"]}}},{"paper":"/paper/gamebench-evaluating-strategic-reasoning","slug":"gamebench-evaluating-strategic-reasoning","title":"GameBench: Evaluating Strategic Reasoning Abilities of LLM Agents","date":"2024-06-07","arxiv_id":"2406.06613","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Joshuaclymer/GameBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"low-resource-cross-lingual-summarization","title":"Low-Resource Cross-Lingual Summarization through Few-Shot Learning with Large Language Models","date":"2024-06-07","arxiv_id":"2406.04630","n_code_links":0,"syntology":null},{"paper":"/paper/horae-a-domain-agnostic-modeling-language-for","slug":"horae-a-domain-agnostic-modeling-language-for","title":"HORAE: A Domain-Agnostic Language for Automated Service Regulation","date":"2024-06-06","arxiv_id":"2406.06600","n_code_links":1,"syntology":null},{"paper":"/paper/llmembed-rethinking-lightweight-llm-s-genuine","slug":"llmembed-rethinking-lightweight-llm-s-genuine","title":"LLMEmbed: Rethinking Lightweight LLM's Genuine Function in Text Classification","date":"2024-06-06","arxiv_id":"2406.03725","n_code_links":1,"syntology":null},{"paper":"/paper/tox-bart-leveraging-toxicity-attributes-for","slug":"tox-bart-leveraging-toxicity-attributes-for","title":"Tox-BART: Leveraging Toxicity Attributes for Explanation Generation of Implicit Hate Speech","date":"2024-06-06","arxiv_id":"2406.03953","n_code_links":1,"syntology":null},{"paper":"/paper/automating-turkish-educational-quiz","slug":"automating-turkish-educational-quiz","title":"Automating Turkish Educational Quiz Generation Using Large Language Models","date":"2024-06-05","arxiv_id":"2406.03397","n_code_links":4,"syntology":null},{"paper":null,"slug":"exploring-multilingual-large-language-models","title":"Exploring Multilingual Large Language Models for Enhanced TNM classification of Radiology Report in lung cancer staging","date":"2024-06-05","arxiv_id":"2406.06591","n_code_links":0,"syntology":null},{"paper":null,"slug":"statbot-swiss-bilingual-open-data-exploration","title":"StatBot.Swiss: Bilingual Open Data Exploration in Natural Language","date":"2024-06-05","arxiv_id":"2406.03170","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-good-the-bad-and-the-hulk-like-gpt","title":"The Good, the Bad, and the Hulk-like GPT: Analyzing Emotional Decisions of Large Language Models in Cooperation and Bargaining Games","date":"2024-06-05","arxiv_id":"2406.03299","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-model-enabled-multi-agent","title":"Large Language Model-Enabled Multi-Agent Manufacturing Systems","date":"2024-06-04","arxiv_id":"2406.01893","n_code_links":0,"syntology":null},{"paper":null,"slug":"luna-an-evaluation-foundation-model-to-catch","title":"Luna: An Evaluation Foundation Model to Catch Language Model Hallucinations with High Accuracy and Low Cost","date":"2024-06-03","arxiv_id":"2406.00975","n_code_links":0,"syntology":null},{"paper":"/paper/semcoder-training-code-language-models-with","slug":"semcoder-training-code-language-models-with","title":"SemCoder: Training Code Language Models with Comprehensive Semantics Reasoning","date":"2024-06-03","arxiv_id":"2406.01006","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":10,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["arise-lab/semcoder"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"superhuman-performance-in-urology-board","title":"Superhuman performance in urology board questions by an explainable large language model enabled for context integration of the European Association of Urology guidelines: the UroBot study","date":"2024-06-03","arxiv_id":"2406.01428","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-distractor-generation-via-large","title":"Unsupervised Distractor Generation via Large Language Model Distilling and Counterfactual Contrastive Decoding","date":"2024-06-03","arxiv_id":"2406.01306","n_code_links":0,"syntology":null},{"paper":null,"slug":"applying-fine-tuned-llms-for-reducing-data","title":"Applying Fine-Tuned LLMs for Reducing Data Needs in Load Profile Analysis","date":"2024-06-02","arxiv_id":"2406.02479","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-mathematical-reasoning-of-large","slug":"evaluating-mathematical-reasoning-of-large","title":"Evaluating Mathematical Reasoning of Large Language Models: A Focus on Error Identification and Correction","date":"2024-06-02","arxiv_id":"2406.00755","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["littlecirc1e/eic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-evaluation-benchmark-for-autoformalization","title":"An Evaluation Benchmark for Autoformalization in Lean4","date":"2024-06-01","arxiv_id":"2406.06555","n_code_links":0,"syntology":null},{"paper":null,"slug":"beyond-metrics-evaluating-llms-effectiveness","title":"Beyond Metrics: Evaluating LLMs' Effectiveness in Culturally Nuanced, Low-Resource Real-World Scenarios","date":"2024-06-01","arxiv_id":"2406.00343","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-zero-shot-next","slug":"large-language-models-are-zero-shot-next","title":"Large Language Models are Zero-Shot Next Location Predictors","date":"2024-05-31","arxiv_id":"2405.20962","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ssai-trento/llm-zero-shot-nl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multilingual-text-style-transfer-datasets","slug":"multilingual-text-style-transfer-datasets","title":"Multilingual Text Style Transfer: Datasets & Models for Indian Languages","date":"2024-05-31","arxiv_id":"2405.20805","n_code_links":2,"syntology":null},{"paper":null,"slug":"the-point-of-view-of-a-sentiment-towards","title":"The Point of View of a Sentiment: Towards Clinician Bias Detection in Psychiatric Notes","date":"2024-05-31","arxiv_id":"2405.20582","n_code_links":0,"syntology":null},{"paper":"/paper/anah-analytical-annotation-of-hallucinations","slug":"anah-analytical-annotation-of-hallucinations","title":"ANAH: Analytical Annotation of Hallucinations in Large Language Models","date":"2024-05-30","arxiv_id":"2405.20315","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["open-compass/anah"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"autobreach-universal-and-adaptive","title":"AutoBreach: Universal and Adaptive Jailbreaking with Efficient Wordplay-Guided Optimization","date":"2024-05-30","arxiv_id":"2405.19668","n_code_links":0,"syntology":null},{"paper":null,"slug":"divide-and-conquer-meets-consensus-unleashing","title":"Divide-and-Conquer Meets Consensus: Unleashing the Power of Functions in Code Generation","date":"2024-05-30","arxiv_id":"2405.20092","n_code_links":0,"syntology":null},{"paper":null,"slug":"phantom-general-trigger-attacks-on-retrieval","title":"Phantom: General Trigger Attacks on Retrieval Augmented Language Generation","date":"2024-05-30","arxiv_id":"2405.20485","n_code_links":0,"syntology":null},{"paper":null,"slug":"robo-instruct-simulator-augmented-instruction","title":"Robo-Instruct: Simulator-Augmented Instruction Alignment For Finetuning Code LLMs","date":"2024-05-30","arxiv_id":"2405.20179","n_code_links":0,"syntology":null},{"paper":"/paper/towards-ontology-enhanced-representation","slug":"towards-ontology-enhanced-representation","title":"Towards Ontology-Enhanced Representation Learning for Large Language Models","date":"2024-05-30","arxiv_id":"2405.20527","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-multi-source-retrieval-question-answering","title":"A Multi-Source Retrieval Question Answering Framework Based on RAG","date":"2024-05-29","arxiv_id":"2405.19207","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-agreement-diagnosing-the-rationale","slug":"beyond-agreement-diagnosing-the-rationale","title":"Beyond Agreement: Diagnosing the Rationale Alignment of Automated Essay Scoring Methods based on Linguistically-informed Counterfactuals","date":"2024-05-29","arxiv_id":"2405.19433","n_code_links":1,"syntology":null},{"paper":"/paper/aligning-to-thousands-of-preferences-via","slug":"aligning-to-thousands-of-preferences-via","title":"Aligning to Thousands of Preferences via System Message Generalization","date":"2024-05-28","arxiv_id":"2405.17977","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kaistAI/Janus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/an-empirical-analysis-on-large-language","slug":"an-empirical-analysis-on-large-language","title":"An Empirical Analysis on Large Language Models in Debate Evaluation","date":"2024-05-28","arxiv_id":"2406.00050","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xinyiliu0227/llm_debate_bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"edinburgh-clinical-nlp-at-mediqa-corr-2024","title":"Edinburgh Clinical NLP at MEDIQA-CORR 2024: Guiding Large Language Models with Hints","date":"2024-05-28","arxiv_id":"2405.18028","n_code_links":0,"syntology":null},{"paper":"/paper/assessing-llms-suitability-for-knowledge","slug":"assessing-llms-suitability-for-knowledge","title":"Assessing LLMs Suitability for Knowledge Graph Completion","date":"2024-05-27","arxiv_id":"2405.17249","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-based-cooperative-agents-using","title":"REVECA: Adaptive Planning and Trajectory-based Validation in Cooperative Language Agents using Information Relevance and Relative Proximity","date":"2024-05-27","arxiv_id":"2405.16751","n_code_links":0,"syntology":null},{"paper":"/paper/reflectioncoder-learning-from-reflection","slug":"reflectioncoder-learning-from-reflection","title":"ReflectionCoder: Learning from Reflection Sequence for Enhanced One-off Code Generation","date":"2024-05-27","arxiv_id":"2405.17057","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["sensellm/reflectioncoder"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/rtl-repo-a-benchmark-for-evaluating-llms-on","slug":"rtl-repo-a-benchmark-for-evaluating-llms-on","title":"RTL-Repo: A Benchmark for Evaluating LLMs on Large-Scale RTL Design Projects","date":"2024-05-27","arxiv_id":"2405.17378","n_code_links":1,"syntology":null},{"paper":"/paper/thread-thinking-deeper-with-recursive","slug":"thread-thinking-deeper-with-recursive","title":"THREAD: Thinking Deeper with Recursive Spawning","date":"2024-05-27","arxiv_id":"2405.17402","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["philipmit/thread"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/automanual-generating-instruction-manuals-by","slug":"automanual-generating-instruction-manuals-by","title":"AutoManual: Constructing Instruction Manuals by LLM Agents via Interactive Environmental Learning","date":"2024-05-25","arxiv_id":"2405.16247","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["minghchen/automanual"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mindstar-enhancing-math-reasoning-in-pre","title":"MindStar: Enhancing Math Reasoning in Pre-trained LLMs at Inference Time","date":"2024-05-25","arxiv_id":"2405.16265","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-evaluation-of-estimative-uncertainty-in","title":"An Evaluation of Estimative Uncertainty in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15185","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-pre-trained-large-language","title":"Benchmarking the Performance of Pre-trained LLMs across Urdu NLP Tasks","date":"2024-05-24","arxiv_id":"2405.15453","n_code_links":0,"syntology":null},{"paper":"/paper/culturepark-boosting-cross-cultural","slug":"culturepark-boosting-cross-cultural","title":"CulturePark: Boosting Cross-cultural Understanding in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15145","n_code_links":1,"syntology":{"ran":0,"of":7,"n_ran_checked":0,"n_instrument":0,"unverified":7,"pointer_only":7,"phrase":"0 ran · 7 unverified","official":{"repos":["scarelette/culturepark"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":[]}}},{"paper":"/paper/evaluating-the-adversarial-robustness-of-1","slug":"evaluating-the-adversarial-robustness-of-1","title":"Evaluating and Safeguarding the Adversarial Robustness of Retrieval-Based In-Context Learning","date":"2024-05-24","arxiv_id":"2405.15984","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["simonucl/adv-retreival-icl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/generalizable-and-scalable-multistage","slug":"generalizable-and-scalable-multistage","title":"Generalizable and Scalable Multistage Biomedical Concept Normalization Leveraging Large Language Models","date":"2024-05-24","arxiv_id":"2405.15122","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-is-not-an-annotator-the-necessity-of","title":"GPT is Not an Annotator: The Necessity of Human Annotation in Fairness Benchmark Construction","date":"2024-05-24","arxiv_id":"2405.15760","n_code_links":0,"syntology":null},{"paper":"/paper/editworld-simulating-world-dynamics-for","slug":"editworld-simulating-world-dynamics-for","title":"EditWorld: Simulating World Dynamics for Instruction-Following Image Editing","date":"2024-05-23","arxiv_id":"2405.14785","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yangling0818/editworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/eliciting-informative-text-evaluations-with","slug":"eliciting-informative-text-evaluations-with","title":"Eliciting Informative Text Evaluations with Large Language Models","date":"2024-05-23","arxiv_id":"2405.15077","n_code_links":1,"syntology":{"ran":10,"of":14,"n_ran_checked":8,"n_instrument":2,"unverified":4,"pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yx-lu/eliciting-informative-text-evaluations-with-large-language-models"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-models-can-self-correct-with","title":"Large Language Models Can Self-Correct with Key Condition Verification","date":"2024-05-23","arxiv_id":"2405.14092","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-large-language-models-with-human","slug":"evaluating-large-language-models-with-human","title":"Evaluating Large Language Models with Human Feedback: Establishing a Swedish Benchmark","date":"2024-05-22","arxiv_id":"2405.14006","n_code_links":1,"syntology":null},{"paper":null,"slug":"ku-dmis-at-ehrsql-2024-generating-sql-query","title":"KU-DMIS at EHRSQL 2024:Generating SQL query via question templatization in EHR","date":"2024-05-22","arxiv_id":"2406.00014","n_code_links":0,"syntology":null},{"paper":"/paper/topa-extend-large-language-models-for-video","slug":"topa-extend-large-language-models-for-video","title":"TOPA: Extending Large Language Models for Video Understanding via Text-Only Pre-Alignment","date":"2024-05-22","arxiv_id":"2405.13911","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":4,"n_instrument":2,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dhg-wei/topa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"generative-ai-and-large-language-models-for","title":"Generative AI in Cybersecurity: A Comprehensive Review of LLM Applications and Vulnerabilities","date":"2024-05-21","arxiv_id":"2405.12750","n_code_links":0,"syntology":null},{"paper":null,"slug":"davinci-at-semeval-2024-task-9-few-shot","title":"DaVinci at SemEval-2024 Task 9: Few-shot prompting GPT-3.5 for Unconventional Reasoning","date":"2024-05-19","arxiv_id":"2405.11559","n_code_links":0,"syntology":null},{"paper":"/paper/zero-shot-stance-detection-using-contextual","slug":"zero-shot-stance-detection-using-contextual","title":"Zero-Shot Stance Detection using Contextual Data Generation with LLMs","date":"2024-05-19","arxiv_id":"2405.11637","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Babakbehkamkia/GPT-Stance-Detection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"67f2fea1445458a09ba5e9bb28ef15706cd53d513da706869bb8a95b64df4bdf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}