{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/7","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":20,"rows_per_page":100,"rows":[601,700],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/6","next":"/method/gpt-3/papers/8","papers":[{"paper":"/paper/ellen-extremely-lightly-supervised-learning","slug":"ellen-extremely-lightly-supervised-learning","title":"ELLEN: Extremely Lightly Supervised Learning For Efficient Named Entity Recognition","date":"2024-03-26","arxiv_id":"2403.17385","n_code_links":1,"syntology":null},{"paper":null,"slug":"enhancing-legal-document-retrieval-a-multi","title":"Enhancing Legal Document Retrieval: A Multi-Phase Approach with Large Language Models","date":"2024-03-26","arxiv_id":"2403.18093","n_code_links":0,"syntology":null},{"paper":null,"slug":"magis-llm-based-multi-agent-framework-for","title":"MAGIS: LLM-Based Multi-Agent Framework for GitHub Issue Resolution","date":"2024-03-26","arxiv_id":"2403.17927","n_code_links":0,"syntology":null},{"paper":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"verbing-weirds-language-models-evaluation-of","title":"Verbing Weirds Language (Models): Evaluation of English Zero-Derivation in Five LLMs","date":"2024-03-26","arxiv_id":"2403.17856","n_code_links":0,"syntology":null},{"paper":"/paper/a-comparison-of-human-gpt-3-5-and-gpt-4","slug":"a-comparison-of-human-gpt-3-5-and-gpt-4","title":"A comparison of Human, GPT-3.5, and GPT-4 Performance in a University-Level Coding Course","date":"2024-03-25","arxiv_id":"2403.16977","n_code_links":1,"syntology":null},{"paper":"/paper/iterative-refinement-of-project-level-code","slug":"iterative-refinement-of-project-level-code","title":"Iterative Refinement of Project-Level Code Context for Precise Code Generation with Compiler Feedback","date":"2024-03-25","arxiv_id":"2403.16792","n_code_links":1,"syntology":null},{"paper":"/paper/repairagent-an-autonomous-llm-based-agent-for","slug":"repairagent-an-autonomous-llm-based-agent-for","title":"RepairAgent: An Autonomous, LLM-Based Agent for Program Repair","date":"2024-03-25","arxiv_id":"2403.17134","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sola-st/RepairAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/textit-linkprompt-natural-and-universal","slug":"textit-linkprompt-natural-and-universal","title":"$\\textit{LinkPrompt}$: Natural and Universal Adversarial Attacks on Prompt-based Language Models","date":"2024-03-25","arxiv_id":"2403.16432","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["savannahxu79/linkprompt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/towards-algorithmic-fidelity-mental-health","slug":"towards-algorithmic-fidelity-mental-health","title":"Towards Algorithmic Fidelity: Mental Health Representation across Demographics in Synthetic vs. Human-generated Data","date":"2024-03-25","arxiv_id":"2403.16909","n_code_links":1,"syntology":null},{"paper":null,"slug":"sql-encoder-improving-nl2sql-in-context","title":"SQL-Encoder: Improving NL2SQL In-Context Learning Through a Context-Aware Encoder","date":"2024-03-24","arxiv_id":"2403.16204","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-large-language-models-for-ontoclean","title":"Using Large Language Models for OntoClean-based Ontology Refinement","date":"2024-03-23","arxiv_id":"2403.15864","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-explore-in-context","title":"Can large language models explore in-context?","date":"2024-03-22","arxiv_id":"2403.15371","n_code_links":0,"syntology":null},{"paper":"/paper/comprehensive-evaluation-and-insights-into-1","slug":"comprehensive-evaluation-and-insights-into-1","title":"Comprehensive Evaluation and Insights into the Use of Large Language Models in the Automation of Behavior-Driven Development Acceptance Test Formulation","date":"2024-03-22","arxiv_id":"2403.14965","n_code_links":1,"syntology":null},{"paper":null,"slug":"text-clustering-with-llm-embeddings","title":"Text Clustering with Large Language Model Embeddings","date":"2024-03-22","arxiv_id":"2403.15112","n_code_links":0,"syntology":null},{"paper":"/paper/vurf-a-general-purpose-reasoning-and-self","slug":"vurf-a-general-purpose-reasoning-and-self","title":"VURF: A General-purpose Reasoning and Self-refinement Framework for Video Understanding","date":"2024-03-21","arxiv_id":"2403.14743","n_code_links":1,"syntology":null},{"paper":"/paper/motion-generation-from-fine-grained-textual","slug":"motion-generation-from-fine-grained-textual","title":"Motion Generation from Fine-grained Textual Descriptions","date":"2024-03-20","arxiv_id":"2403.13518","n_code_links":1,"syntology":null},{"paper":null,"slug":"natural-language-as-polices-reasoning-for","title":"Natural Language as Policies: Reasoning for Coordinate-Level Embodied Control with LLMs","date":"2024-03-20","arxiv_id":"2403.13801","n_code_links":0,"syntology":null},{"paper":null,"slug":"paramanu-ayn-an-efficient-novel-generative","title":"PARAMANU-AYN: Pretrain from scratch or Continual Pretraining of LLMs for Legal Domain Adaptation?","date":"2024-03-20","arxiv_id":"2403.13681","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-data-curation-for-robust-language","title":"Automated Data Curation for Robust Language Model Fine-Tuning","date":"2024-03-19","arxiv_id":"2403.12776","n_code_links":0,"syntology":null},{"paper":"/paper/fine-tuning-pre-trained-language-models-to","slug":"fine-tuning-pre-trained-language-models-to","title":"Fine-Tuning Pre-trained Language Models to Detect In-Game Trash Talks","date":"2024-03-19","arxiv_id":"2403.15458","n_code_links":0,"syntology":null},{"paper":"/paper/instructing-large-language-models-to-identify","slug":"instructing-large-language-models-to-identify","title":"Instructing Large Language Models to Identify and Ignore Irrelevant Conditions","date":"2024-03-19","arxiv_id":"2403.12744","n_code_links":1,"syntology":null},{"paper":null,"slug":"construction-of-hyper-relational-knowledge","title":"Construction of Hyper-Relational Knowledge Graphs Using Pre-Trained Large Language Models","date":"2024-03-18","arxiv_id":"2403.11786","n_code_links":0,"syntology":null},{"paper":"/paper/easyjailbreak-a-unified-framework-for","slug":"easyjailbreak-a-unified-framework-for","title":"EasyJailbreak: A Unified Framework for Jailbreaking Large Language Models","date":"2024-03-18","arxiv_id":"2403.12171","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["easyjailbreak/easyjailbreak"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ensuring-safe-and-high-quality-outputs-a","slug":"ensuring-safe-and-high-quality-outputs-a","title":"Ensuring Safe and High-Quality Outputs: A Guideline Library Approach for Language Models","date":"2024-03-18","arxiv_id":"2403.11838","n_code_links":1,"syntology":null},{"paper":null,"slug":"gpt-4-as-evaluator-evaluating-large-language","title":"GPT-4 as Evaluator: Evaluating Large Language Models on Pest Management in Agriculture","date":"2024-03-18","arxiv_id":"2403.11858","n_code_links":0,"syntology":null},{"paper":"/paper/hatecot-an-explanation-enhanced-dataset-for","slug":"hatecot-an-explanation-enhanced-dataset-for","title":"HateCOT: An Explanation-Enhanced Dataset for Generalizable Offensive Speech Detection via Large Language Models","date":"2024-03-18","arxiv_id":"2403.11456","n_code_links":1,"syntology":null},{"paper":"/paper/how-far-are-we-on-the-decision-making-of-llms","slug":"how-far-are-we-on-the-decision-making-of-llms","title":"How Far Are We on the Decision-Making of LLMs? Evaluating LLMs' Gaming Ability in Multi-Agent Environments","date":"2024-03-18","arxiv_id":"2403.11807","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cuhk-arise/gamabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"metaphor-understanding-challenge-dataset-for","title":"Metaphor Understanding Challenge Dataset for LLMs","date":"2024-03-18","arxiv_id":"2403.11810","n_code_links":0,"syntology":null},{"paper":null,"slug":"shifting-the-lens-detecting-malware-in-npm","title":"Leveraging Large Language Models to Detect npm Malicious Packages","date":"2024-03-18","arxiv_id":"2403.12196","n_code_links":0,"syntology":null},{"paper":"/paper/data-is-all-you-need-finetuning-llms-for-chip","slug":"data-is-all-you-need-finetuning-llms-for-chip","title":"Data is all you need: Finetuning LLMs for Chip Design via an Automated design-data augmentation framework","date":"2024-03-17","arxiv_id":"2403.11202","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aichipdesign/chipgptft"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"large-language-model-powered-chatbots-for","title":"Large language model-powered chatbots for internationalizing student support in higher education","date":"2024-03-16","arxiv_id":"2403.14702","n_code_links":0,"syntology":null},{"paper":null,"slug":"exegpt-constraint-aware-resource-scheduling","title":"ExeGPT: Constraint-Aware Resource Scheduling for LLM Inference","date":"2024-03-15","arxiv_id":"2404.07947","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-condensation-and-reasoning-for","title":"Knowledge Condensation and Reasoning for Knowledge-based VQA","date":"2024-03-15","arxiv_id":"2403.10037","n_code_links":0,"syntology":null},{"paper":"/paper/codeultrafeedback-an-llm-as-a-judge-dataset","slug":"codeultrafeedback-an-llm-as-a-judge-dataset","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","date":"2024-03-14","arxiv_id":"2403.09032","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["martin-wey/codeultrafeedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-llms-for-gender-disparities-in","title":"Evaluating LLMs for Gender Disparities in Notable Persons","date":"2024-03-14","arxiv_id":"2403.09148","n_code_links":0,"syntology":null},{"paper":null,"slug":"komodo-a-linguistic-expedition-into-indonesia","title":"Komodo: A Linguistic Expedition into Indonesia's Regional Languages","date":"2024-03-14","arxiv_id":"2403.09362","n_code_links":0,"syntology":null},{"paper":null,"slug":"sabia-2-a-new-generation-of-portuguese-large","title":"Sabiá-2: A New Generation of Portuguese Large Language Models","date":"2024-03-14","arxiv_id":"2403.09887","n_code_links":0,"syntology":null},{"paper":null,"slug":"rethinking-generative-large-language-model","title":"Rethinking Generative Large Language Model Evaluation for Semantic Comprehension","date":"2024-03-12","arxiv_id":"2403.07872","n_code_links":0,"syntology":null},{"paper":null,"slug":"sifid-reassess-summary-factual-inconsistency","title":"SIFiD: Reassess Summary Factual Inconsistency Detection with LLM","date":"2024-03-12","arxiv_id":"2403.07557","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-future-of-document-indexing-gpt-and-donut","title":"The future of document indexing: GPT and Donut revolutionize table of content processing","date":"2024-03-12","arxiv_id":"2403.07553","n_code_links":0,"syntology":null},{"paper":null,"slug":"development-of-a-reliable-and-accessible","title":"Development of a Reliable and Accessible Caregiving Language Model (CaLM)","date":"2024-03-11","arxiv_id":"2403.06857","n_code_links":0,"syntology":null},{"paper":null,"slug":"hybrid-human-llm-corpus-construction-and-llm","title":"Hybrid Human-LLM Corpus Construction and LLM Evaluation for Rare Linguistic Phenomena","date":"2024-03-11","arxiv_id":"2403.06965","n_code_links":0,"syntology":null},{"paper":null,"slug":"narrating-causal-graphs-with-large-language","title":"Narrating Causal Graphs with Large Language Models","date":"2024-03-11","arxiv_id":"2403.07118","n_code_links":0,"syntology":null},{"paper":"/paper/bias-augmented-consistency-training-reduces","slug":"bias-augmented-consistency-training-reduces","title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","date":"2024-03-08","arxiv_id":"2403.05518","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["raybears/cot-transparency"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rat-retrieval-augmented-thoughts-elicit","slug":"rat-retrieval-augmented-thoughts-elicit","title":"RAT: Retrieval Augmented Thoughts Elicit Context-Aware Reasoning in Long-Horizon Generation","date":"2024-03-08","arxiv_id":"2403.05313","n_code_links":1,"syntology":null},{"paper":null,"slug":"feedback-generation-for-programming-exercises","title":"Feedback-Generation for Programming Exercises With GPT-4","date":"2024-03-07","arxiv_id":"2403.04449","n_code_links":0,"syntology":null},{"paper":null,"slug":"telecom-language-models-must-they-be-large","title":"Telecom Language Models: Must They Be Large?","date":"2024-03-07","arxiv_id":"2403.04666","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-hallucination-in-large-language","slug":"benchmarking-hallucination-in-large-language","title":"Benchmarking Hallucination in Large Language Models based on Unanswerable Math Word Problem","date":"2024-03-06","arxiv_id":"2403.03558","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-large-language-models-do-analytical","title":"Can Large Language Models do Analytical Reasoning?","date":"2024-03-06","arxiv_id":"2403.04031","n_code_links":0,"syntology":null},{"paper":null,"slug":"general2specialized-llms-translation-for-e","title":"General2Specialized LLMs Translation for E-commerce","date":"2024-03-06","arxiv_id":"2403.03689","n_code_links":0,"syntology":null},{"paper":null,"slug":"guiding-enumerative-program-synthesis-with","title":"Guiding Enumerative Program Synthesis with Large Language Models","date":"2024-03-06","arxiv_id":"2403.03997","n_code_links":0,"syntology":null},{"paper":"/paper/rapidly-developing-high-quality-instruction","slug":"rapidly-developing-high-quality-instruction","title":"Rapidly Developing High-quality Instruction Data and Evaluation Benchmark for Large Language Models with Minimal Human Effort: A Case Study on Japanese","date":"2024-03-06","arxiv_id":"2403.03690","n_code_links":2,"syntology":null},{"paper":"/paper/evaluating-and-optimizing-educational-content","slug":"evaluating-and-optimizing-educational-content","title":"Evaluating and Optimizing Educational Content with Large Language Model Judgments","date":"2024-03-05","arxiv_id":"2403.02795","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["StanfordAI4HI/ed-expert-simulator"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"improving-event-definition-following-for-zero","title":"Improving Event Definition Following For Zero-Shot Event Detection","date":"2024-03-05","arxiv_id":"2403.02586","n_code_links":0,"syntology":null},{"paper":"/paper/mathscale-scaling-instruction-tuning-for","slug":"mathscale-scaling-instruction-tuning-for","title":"MathScale: Scaling Instruction Tuning for Mathematical Reasoning","date":"2024-03-05","arxiv_id":"2403.02884","n_code_links":1,"syntology":null},{"paper":null,"slug":"privacy-aware-semantic-cache-for-large","title":"MeanCache: User-Centric Semantic Caching for LLM Web Services","date":"2024-03-05","arxiv_id":"2403.02694","n_code_links":0,"syntology":null},{"paper":null,"slug":"zero-shot-cross-lingual-document-level-event","title":"Zero-Shot Cross-Lingual Document-Level Event Causality Identification with Heterogeneous Graph Contrastive Transfer Learning","date":"2024-03-05","arxiv_id":"2403.02893","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-generate-architectural-design","title":"Can LLMs Generate Architectural Design Decisions? -An Exploratory Empirical study","date":"2024-03-04","arxiv_id":"2403.01709","n_code_links":0,"syntology":null},{"paper":"/paper/differentially-private-synthetic-data-via-1","slug":"differentially-private-synthetic-data-via-1","title":"Differentially Private Synthetic Data via Foundation Model APIs 2: Text","date":"2024-03-04","arxiv_id":"2403.01749","n_code_links":2,"syntology":{"ran":6,"of":11,"n_ran_checked":6,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["ai-secure/aug-pe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hypertext-entity-extraction-in-webpage","title":"Hypertext Entity Extraction in Webpage","date":"2024-03-04","arxiv_id":"2403.01698","n_code_links":0,"syntology":null},{"paper":null,"slug":"phantom-personality-has-an-effect-on-theory","title":"PHAnToM: Persona-based Prompting Has An Effect on Theory-of-Mind Reasoning in Large Language Models","date":"2024-03-04","arxiv_id":"2403.02246","n_code_links":0,"syntology":null},{"paper":"/paper/protrix-building-models-for-planning-and","slug":"protrix-building-models-for-planning-and","title":"ProTrix: Building Models for Planning and Reasoning over Tables with Sentence Context","date":"2024-03-04","arxiv_id":"2403.02177","n_code_links":1,"syntology":null},{"paper":"/paper/sciassess-benchmarking-llm-proficiency-in","slug":"sciassess-benchmarking-llm-proficiency-in","title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","date":"2024-03-04","arxiv_id":"2403.01976","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sci-assess/sciassess"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/using-llms-for-the-extraction-and","slug":"using-llms-for-the-extraction-and","title":"Using LLMs for the Extraction and Normalization of Product Attribute Values","date":"2024-03-04","arxiv_id":"2403.02130","n_code_links":1,"syntology":null},{"paper":"/paper/serval-synergy-learning-between-vertical","slug":"serval-synergy-learning-between-vertical","title":"SERVAL: Synergy Learning between Vertical Models and LLMs towards Oracle-Level Zero-shot Medical Prediction","date":"2024-03-03","arxiv_id":"2403.01570","n_code_links":0,"syntology":{"ran":6,"of":8,"n_ran_checked":1,"n_instrument":5,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/autodefense-multi-agent-llm-defense-against","slug":"autodefense-multi-agent-llm-defense-against","title":"AutoDefense: Multi-Agent LLM Defense against Jailbreak Attacks","date":"2024-03-02","arxiv_id":"2403.04783","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xhmy/autodefense"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lm4opt-unveiling-the-potential-of-large","title":"LM4OPT: Unveiling the Potential of Large Language Models in Formulating Mathematical Optimization Problems","date":"2024-03-02","arxiv_id":"2403.01342","n_code_links":0,"syntology":null},{"paper":"/paper/softtiger-a-clinical-foundation-model-for","slug":"softtiger-a-clinical-foundation-model-for","title":"SoftTiger: A Clinical Foundation Model for Healthcare Workflows","date":"2024-03-01","arxiv_id":"2403.00868","n_code_links":1,"syntology":null},{"paper":"/paper/artist-automated-text-simplification-for-task","slug":"artist-automated-text-simplification-for-task","title":"ARTiST: Automated Text Simplification for Task Guidance in Augmented Reality","date":"2024-02-29","arxiv_id":"2402.18797","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-ensemble-optimal-large-language-model","title":"LLM-Ensemble: Optimal Large Language Model Ensemble Method for E-commerce Product Attribute Value Extraction","date":"2024-02-29","arxiv_id":"2403.00863","n_code_links":0,"syntology":null},{"paper":null,"slug":"proc2pddl-open-domain-planning","title":"PROC2PDDL: Open-Domain Planning Representations from Texts","date":"2024-02-29","arxiv_id":"2403.00092","n_code_links":0,"syntology":null},{"paper":null,"slug":"vixen-visual-text-comparison-network-for","title":"VIXEN: Visual Text Comparison Network for Image Difference Captioning","date":"2024-02-29","arxiv_id":"2402.19119","n_code_links":0,"syntology":null},{"paper":null,"slug":"decomposed-prompting-unveiling-multilingual","title":"Decomposed Prompting: Unveiling Multilingual Linguistic Structure Knowledge in English-Centric Large Language Models","date":"2024-02-28","arxiv_id":"2402.18397","n_code_links":0,"syntology":null},{"paper":"/paper/keeping-llms-aligned-after-fine-tuning-the","slug":"keeping-llms-aligned-after-fine-tuning-the","title":"Keeping LLMs Aligned After Fine-tuning: The Crucial Role of Prompt Templates","date":"2024-02-28","arxiv_id":"2402.18540","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":5,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["vfleaking/ptst"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-learning-detection-method-for-large","title":"Deep Learning Detection Method for Large Language Models-Generated Scientific Content","date":"2024-02-27","arxiv_id":"2403.00828","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-vision-language-stem-skills-of","slug":"measuring-vision-language-stem-skills-of","title":"Measuring Vision-Language STEM Skills of Neural Models","date":"2024-02-27","arxiv_id":"2402.17205","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["stemdataset/STEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatmusician-understanding-and-generating","slug":"chatmusician-understanding-and-generating","title":"ChatMusician: Understanding and Generating Music Intrinsically with LLM","date":"2024-02-25","arxiv_id":"2402.16153","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hf-lin/ChatMusician"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"deep-learning-approaches-for-improving","title":"Deep Learning Approaches for Improving Question Answering Systems in Hepatocellular Carcinoma Research","date":"2024-02-25","arxiv_id":"2402.16038","n_code_links":0,"syntology":null},{"paper":"/paper/fusechat-knowledge-fusion-of-chat-models","slug":"fusechat-knowledge-fusion-of-chat-models","title":"Knowledge Fusion of Chat LLMs: A Preliminary Technical Report","date":"2024-02-25","arxiv_id":"2402.16107","n_code_links":2,"syntology":null},{"paper":null,"slug":"llms-can-defend-themselves-against","title":"LLMs Can Defend Themselves Against Jailbreaking in a Practical Manner: A Vision Paper","date":"2024-02-24","arxiv_id":"2402.15727","n_code_links":0,"syntology":null},{"paper":null,"slug":"look-before-you-leap-problem-elaboration","title":"Look Before You Leap: Problem Elaboration Prompting Improves Mathematical Reasoning in Large Language Models","date":"2024-02-24","arxiv_id":"2402.15764","n_code_links":0,"syntology":null},{"paper":"/paper/attributionbench-how-hard-is-automatic","slug":"attributionbench-how-hard-is-automatic","title":"AttributionBench: How Hard is Automatic Attribution Evaluation?","date":"2024-02-23","arxiv_id":"2402.15089","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-large-language-models-for-domain","title":"Fine-tuning Large Language Models for Domain-specific Machine Translation","date":"2024-02-23","arxiv_id":"2402.15061","n_code_links":0,"syntology":null},{"paper":"/paper/prompting-llms-to-compose-meta-review-drafts","slug":"prompting-llms-to-compose-meta-review-drafts","title":"LLMs as Meta-Reviewers' Assistants: A Case Study","date":"2024-02-23","arxiv_id":"2402.15589","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-large-language-models-detect","title":"Can Large Language Models Detect Misinformation in Scientific News Reporting?","date":"2024-02-22","arxiv_id":"2402.14268","n_code_links":0,"syntology":null},{"paper":null,"slug":"copilot-evaluation-harness-evaluating-llm","title":"Copilot Evaluation Harness: Evaluating LLM-Guided Software Programming","date":"2024-02-22","arxiv_id":"2402.14261","n_code_links":0,"syntology":null},{"paper":"/paper/hint-before-solving-prompting-guiding-llms-to","slug":"hint-before-solving-prompting-guiding-llms-to","title":"Hint-before-Solving Prompting: Guiding LLMs to Effectively Utilize Encoded Knowledge","date":"2024-02-22","arxiv_id":"2402.14310","n_code_links":1,"syntology":null},{"paper":"/paper/kocosa-korean-context-aware-sarcasm-detection","slug":"kocosa-korean-context-aware-sarcasm-detection","title":"KoCoSa: Korean Context-aware Sarcasm Detection Dataset","date":"2024-02-22","arxiv_id":"2402.14428","n_code_links":1,"syntology":null},{"paper":null,"slug":"roboscript-code-generation-for-free-form","title":"RoboScript: Code Generation for Free-Form Manipulation Tasks across Real and Simulation","date":"2024-02-22","arxiv_id":"2402.14623","n_code_links":0,"syntology":null},{"paper":"/paper/tokenization-counts-the-impact-of","slug":"tokenization-counts-the-impact-of","title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","date":"2024-02-22","arxiv_id":"2402.14903","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aadityasingh/tokenizationcounts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"beyond-hate-speech-nlp-s-challenges-and","title":"Beyond Hate Speech: NLP's Challenges and Opportunities in Uncovering Dehumanizing Language","date":"2024-02-21","arxiv_id":"2402.13818","n_code_links":0,"syntology":null},{"paper":"/paper/llm-jailbreak-attack-versus-defense","slug":"llm-jailbreak-attack-versus-defense","title":"A Comprehensive Study of Jailbreak Attack versus Defense for Large Language Models","date":"2024-02-21","arxiv_id":"2402.13457","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ltroin/llm_attack_defense_arena"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/synfac-edit-synthetic-imitation-edit-feedback","slug":"synfac-edit-synthetic-imitation-edit-feedback","title":"SYNFAC-EDIT: Synthetic Imitation Edit Feedback for Factual Alignment in Clinical Summarization","date":"2024-02-21","arxiv_id":"2402.13919","n_code_links":1,"syntology":null},{"paper":"/paper/benchmarking-retrieval-augmented-generation","slug":"benchmarking-retrieval-augmented-generation","title":"Benchmarking Retrieval-Augmented Generation for Medicine","date":"2024-02-20","arxiv_id":"2402.13178","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["teddy-xionggz/medrag","teddy-xionggz/mirage"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evograd-a-dynamic-take-on-the-winograd-schema","title":"EvoGrad: A Dynamic Take on the Winograd Schema Challenge with Human Adversaries","date":"2024-02-20","arxiv_id":"2402.13372","n_code_links":0,"syntology":null},{"paper":"/paper/moelora-contrastive-learning-guided-mixture","slug":"moelora-contrastive-learning-guided-mixture","title":"MoELoRA: Contrastive Learning Guided Mixture of Experts on Parameter-Efficient Fine-Tuning for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12851","n_code_links":1,"syntology":null},{"paper":null,"slug":"nl2formula-generating-spreadsheet-formulas","title":"NL2Formula: Generating Spreadsheet Formulas from Natural Language Queries","date":"2024-02-20","arxiv_id":"2402.14853","n_code_links":0,"syntology":null},{"paper":null,"slug":"sql-craft-text-to-sql-through-interactive","title":"$R^3$: \"This is My SQL, Are You With Me?\" A Consensus-Based Multi-Agent System for Text-to-SQL Tasks","date":"2024-02-20","arxiv_id":"2402.14851","n_code_links":0,"syntology":null},{"paper":"/paper/the-impact-of-demonstrations-on-multilingual","slug":"the-impact-of-demonstrations-on-multilingual","title":"The Impact of Demonstrations on Multilingual In-Context Learning: A Multidimensional Analysis","date":"2024-02-20","arxiv_id":"2402.12976","n_code_links":1,"syntology":null}],"record_sha256":"0ee6ac94daaf3312e9716ef3320e427a839d9aeade0738f73b7393b79d8c8b42","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}