{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/16","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":16,"pages_in_order":29,"rows_per_page":100,"rows":[1501,1600],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/15","next":"/method/gpt-4/papers/17","papers":[{"paper":null,"slug":"magis-llm-based-multi-agent-framework-for","title":"MAGIS: LLM-Based Multi-Agent Framework for GitHub Issue Resolution","date":"2024-03-26","arxiv_id":"2403.17927","n_code_links":0,"syntology":null},{"paper":null,"slug":"supervisory-prompt-training","title":"Supervisory Prompt Training","date":"2024-03-26","arxiv_id":"2403.18051","n_code_links":0,"syntology":null},{"paper":null,"slug":"verbing-weirds-language-models-evaluation-of","title":"Verbing Weirds Language (Models): Evaluation of English Zero-Derivation in Five LLMs","date":"2024-03-26","arxiv_id":"2403.17856","n_code_links":0,"syntology":null},{"paper":"/paper/a-comparison-of-human-gpt-3-5-and-gpt-4","slug":"a-comparison-of-human-gpt-3-5-and-gpt-4","title":"A comparison of Human, GPT-3.5, and GPT-4 Performance in a University-Level Coding Course","date":"2024-03-25","arxiv_id":"2403.16977","n_code_links":1,"syntology":null},{"paper":null,"slug":"do-llm-agents-have-regret-a-case-study-in","title":"Do LLM Agents Have Regret? A Case Study in Online Learning and Games","date":"2024-03-25","arxiv_id":"2403.16843","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-understands-discourse-at-least-as-well","title":"Text Understanding in GPT-4 vs Humans","date":"2024-03-25","arxiv_id":"2403.17196","n_code_links":0,"syntology":null},{"paper":"/paper/linear-cross-document-event-coreference","slug":"linear-cross-document-event-coreference","title":"Linear Cross-document Event Coreference Resolution with X-AMR","date":"2024-03-25","arxiv_id":"2404.08656","n_code_links":1,"syntology":null},{"paper":"/paper/sesame-a-framework-to-simulate-self-reported","slug":"sesame-a-framework-to-simulate-self-reported","title":"SeSaMe: A Framework to Simulate Self-Reported Ground Truth for Mental Health Sensing Studies","date":"2024-03-25","arxiv_id":"2403.17219","n_code_links":1,"syntology":null},{"paper":"/paper/state-space-models-as-foundation-models-a","slug":"state-space-models-as-foundation-models-a","title":"State Space Models as Foundation Models: A Control Theoretic Overview","date":"2024-03-25","arxiv_id":"2403.16899","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":5,"n_instrument":1,"unverified":3,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["jsie7/ssm-benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"eagle-a-domain-generalization-framework-for","title":"EAGLE: A Domain Generalization Framework for AI-generated Text Detection","date":"2024-03-23","arxiv_id":"2403.15690","n_code_links":0,"syntology":null},{"paper":"/paper/llambert-large-scale-low-cost-data-annotation","slug":"llambert-large-scale-low-cost-data-annotation","title":"LlamBERT: Large-scale low-cost data annotation in NLP","date":"2024-03-23","arxiv_id":"2403.15938","n_code_links":1,"syntology":null},{"paper":null,"slug":"using-large-language-models-for-ontoclean","title":"Using Large Language Models for OntoClean-based Ontology Refinement","date":"2024-03-23","arxiv_id":"2403.15864","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-explore-in-context","title":"Can large language models explore in-context?","date":"2024-03-22","arxiv_id":"2403.15371","n_code_links":0,"syntology":null},{"paper":"/paper/comprehensive-evaluation-and-insights-into-1","slug":"comprehensive-evaluation-and-insights-into-1","title":"Comprehensive Evaluation and Insights into the Use of Large Language Models in the Automation of Behavior-Driven Development Acceptance Test Formulation","date":"2024-03-22","arxiv_id":"2403.14965","n_code_links":1,"syntology":null},{"paper":"/paper/construction-of-a-japanese-financial","slug":"construction-of-a-japanese-financial","title":"Construction of a Japanese Financial Benchmark for Large Language Models","date":"2024-03-22","arxiv_id":"2403.15062","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pfnet-research/japanese-lm-fin-harness"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"esg-classification-by-implicit-rule-learning","title":"ESG Classification by Implicit Rule Learning via GPT-4","date":"2024-03-22","arxiv_id":"2403.15040","n_code_links":0,"syntology":null},{"paper":null,"slug":"selectively-informative-description-can","title":"Selectively Informative Description can Reduce Undesired Embedding Entanglements in Text-to-Image Personalization","date":"2024-03-22","arxiv_id":"2403.15330","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-chain-of-thought-prompting-approach-with","title":"A Chain-of-Thought Prompting Approach with LLMs for Evaluating Students' Formative Assessment Responses in Science","date":"2024-03-21","arxiv_id":"2403.14565","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-utility-of-large-language","title":"Assessing the Utility of Large Language Models for Phenotype-Driven Gene Prioritization in Rare Genetic Disorder Diagnosis","date":"2024-03-21","arxiv_id":"2403.14801","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-potential-of-large-language-7","title":"Exploring the Potential of Large Language Models in Graph Generation","date":"2024-03-21","arxiv_id":"2403.14358","n_code_links":0,"syntology":null},{"paper":"/paper/k-act2emo-korean-commonsense-knowledge-graph","slug":"k-act2emo-korean-commonsense-knowledge-graph","title":"K-Act2Emo: Korean Commonsense Knowledge Graph for Indirect Emotional Expression","date":"2024-03-21","arxiv_id":"2403.14253","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-based-extraction-of-contradictions-from","title":"LLM-based Extraction of Contradictions from Patents","date":"2024-03-21","arxiv_id":"2403.14258","n_code_links":0,"syntology":null},{"paper":null,"slug":"react-meets-actre-autonomous-annotations-of","title":"ReAct Meets ActRe: When Language Agents Enjoy Training Data Autonomy","date":"2024-03-21","arxiv_id":"2403.14589","n_code_links":0,"syntology":null},{"paper":"/paper/facilitating-pornographic-text-detection-for","slug":"facilitating-pornographic-text-detection-for","title":"Facilitating Pornographic Text Detection for Open-Domain Dialogue Systems via Knowledge Distillation of Large Language Models","date":"2024-03-20","arxiv_id":"2403.13250","n_code_links":1,"syntology":null},{"paper":null,"slug":"natural-language-as-polices-reasoning-for","title":"Natural Language as Policies: Reasoning for Coordinate-Level Embodied Control with LLMs","date":"2024-03-20","arxiv_id":"2403.13801","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-data-curation-for-robust-language","title":"Automated Data Curation for Robust Language Model Fine-Tuning","date":"2024-03-19","arxiv_id":"2403.12776","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-information-extraction-from","title":"Automatic Information Extraction From Employment Tribunal Judgements Using Large Language Models","date":"2024-03-19","arxiv_id":"2403.12936","n_code_links":0,"syntology":null},{"paper":"/paper/encode-once-and-decode-in-parallel-efficient","slug":"encode-once-and-decode-in-parallel-efficient","title":"Efficient Encoder-Decoder Transformer Decoding for Decomposable Tasks","date":"2024-03-19","arxiv_id":"2403.13112","n_code_links":1,"syntology":null},{"paper":"/paper/insight-end-to-end-neuro-symbolic-visual","slug":"insight-end-to-end-neuro-symbolic-visual","title":"End-to-End Neuro-Symbolic Reinforcement Learning with Textual Explanations","date":"2024-03-19","arxiv_id":"2403.12451","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["liruiluo/nsrl-vision-pub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lhmke-a-large-scale-holistic-multi-subject","title":"LHMKE: A Large-scale Holistic Multi-subject Knowledge Evaluation Benchmark for Chinese Large Language Models","date":"2024-03-19","arxiv_id":"2403.12601","n_code_links":0,"syntology":null},{"paper":"/paper/pragmatic-competence-evaluation-of-large","slug":"pragmatic-competence-evaluation-of-large","title":"Pragmatic Competence Evaluation of Large Language Models for the Korean Language","date":"2024-03-19","arxiv_id":"2403.12675","n_code_links":1,"syntology":null},{"paper":null,"slug":"rankprompt-step-by-step-comparisons-make","title":"RankPrompt: Step-by-Step Comparisons Make Language Models Better Reasoners","date":"2024-03-19","arxiv_id":"2403.12373","n_code_links":0,"syntology":null},{"paper":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/counting-stars-a-simple-efficient-and","slug":"counting-stars-a-simple-efficient-and","title":"Counting-Stars: A Multi-evidence, Position-aware, and Scalable Benchmark for Evaluating Long-Context Large Language Models","date":"2024-03-18","arxiv_id":"2403.11802","n_code_links":1,"syntology":null},{"paper":"/paper/easyjailbreak-a-unified-framework-for","slug":"easyjailbreak-a-unified-framework-for","title":"EasyJailbreak: A Unified Framework for Jailbreaking Large Language Models","date":"2024-03-18","arxiv_id":"2403.12171","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["easyjailbreak/easyjailbreak"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-hokkien-dual-translation-by","slug":"enhancing-hokkien-dual-translation-by","title":"Enhancing Taiwanese Hokkien Dual Translation by Exploring and Standardizing of Four Writing Systems","date":"2024-03-18","arxiv_id":"2403.12024","n_code_links":1,"syntology":null},{"paper":"/paper/ensuring-safe-and-high-quality-outputs-a","slug":"ensuring-safe-and-high-quality-outputs-a","title":"Ensuring Safe and High-Quality Outputs: A Guideline Library Approach for Language Models","date":"2024-03-18","arxiv_id":"2403.11838","n_code_links":1,"syntology":null},{"paper":null,"slug":"envgen-generating-and-adapting-environments","title":"EnvGen: Generating and Adapting Environments via LLMs for Training Embodied Agents","date":"2024-03-18","arxiv_id":"2403.12014","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-as-evaluator-evaluating-large-language","title":"GPT-4 as Evaluator: Evaluating Large Language Models on Pest Management in Agriculture","date":"2024-03-18","arxiv_id":"2403.11858","n_code_links":0,"syntology":null},{"paper":"/paper/how-far-are-we-on-the-decision-making-of-llms","slug":"how-far-are-we-on-the-decision-making-of-llms","title":"How Far Are We on the Decision-Making of LLMs? Evaluating LLMs' Gaming Ability in Multi-Agent Environments","date":"2024-03-18","arxiv_id":"2403.11807","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cuhk-arise/gamabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"shifting-the-lens-detecting-malware-in-npm","title":"Leveraging Large Language Models to Detect npm Malicious Packages","date":"2024-03-18","arxiv_id":"2403.12196","n_code_links":0,"syntology":null},{"paper":null,"slug":"aligning-uncertainty-leveraging-llms-to","title":"Aligning Uncertainty: Leveraging LLMs to Analyze Uncertainty Transfer in Text Summarization","date":"2024-03-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/correcting-misinformation-on-social-media","slug":"correcting-misinformation-on-social-media","title":"Correcting misinformation on social media with a large language model","date":"2024-03-17","arxiv_id":"2403.11169","n_code_links":1,"syntology":null},{"paper":null,"slug":"humsum-a-personalized-lecture-summarization","title":"HumSum: A Personalized Lecture Summarization Tool for Humanities Students Using LLMs","date":"2024-03-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/from-words-to-routes-applying-large-language","slug":"from-words-to-routes-applying-large-language","title":"Can Large Language Models Solve Robot Routing?","date":"2024-03-16","arxiv_id":"2403.10795","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-model-powered-chatbots-for","title":"Large language model-powered chatbots for internationalizing student support in higher education","date":"2024-03-16","arxiv_id":"2403.14702","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-thorough-comparison-of-cross-encoders-and","title":"A Thorough Comparison of Cross-Encoders and LLMs for Reranking SPLADE","date":"2024-03-15","arxiv_id":"2403.10407","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-continued-pretrained-llm-approach-for","title":"A Continued Pretrained LLM Approach for Automatic Medical Note Generation","date":"2024-03-14","arxiv_id":"2403.09057","n_code_links":0,"syntology":null},{"paper":null,"slug":"aratrust-an-evaluation-of-trustworthiness-for","title":"AraTrust: An Evaluation of Trustworthiness for LLMs in Arabic","date":"2024-03-14","arxiv_id":"2403.09017","n_code_links":0,"syntology":null},{"paper":"/paper/codeultrafeedback-an-llm-as-a-judge-dataset","slug":"codeultrafeedback-an-llm-as-a-judge-dataset","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","date":"2024-03-14","arxiv_id":"2403.09032","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["martin-wey/codeultrafeedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-llms-for-gender-disparities-in","title":"Evaluating LLMs for Gender Disparities in Notable Persons","date":"2024-03-14","arxiv_id":"2403.09148","n_code_links":0,"syntology":null},{"paper":"/paper/lamp-a-language-model-on-the-map","slug":"lamp-a-language-model-on-the-map","title":"LAMP: A Language Model on the Map","date":"2024-03-14","arxiv_id":"2403.09059","n_code_links":1,"syntology":null},{"paper":null,"slug":"sabia-2-a-new-generation-of-portuguese-large","title":"Sabiá-2: A New Generation of Portuguese Large Language Models","date":"2024-03-14","arxiv_id":"2403.09887","n_code_links":0,"syntology":null},{"paper":null,"slug":"visiongpt-3d-a-generalized-multimodal-agent","title":"VisionGPT-3D: A Generalized Multimodal Agent for Enhanced 3D Vision Understanding","date":"2024-03-14","arxiv_id":"2403.09530","n_code_links":0,"syntology":null},{"paper":"/paper/can-large-language-models-identify-authorship","slug":"can-large-language-models-identify-authorship","title":"Can Large Language Models Identify Authorship?","date":"2024-03-13","arxiv_id":"2403.08213","n_code_links":1,"syntology":null},{"paper":"/paper/devbench-a-comprehensive-benchmark-for","slug":"devbench-a-comprehensive-benchmark-for","title":"Prompting Large Language Models to Tackle the Full Software Development Lifecycle: A Case Study","date":"2024-03-13","arxiv_id":"2403.08604","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["open-compass/devbench","open-compass/deveval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"distilling-named-entity-recognition-models","title":"Distilling Named Entity Recognition Models for Endangered Species from Large Language Models","date":"2024-03-13","arxiv_id":"2403.15430","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-application-of-large-language","title":"Evaluating the Application of Large Language Models to Generate Feedback in Programming Education","date":"2024-03-13","arxiv_id":"2403.09744","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-are-contrastive","slug":"large-language-models-are-contrastive","title":"Large Language Models are Contrastive Reasoners","date":"2024-03-13","arxiv_id":"2403.08211","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-safety-generalization-challenges-of","slug":"exploring-safety-generalization-challenges-of","title":"CodeAttack: Revealing Safety Generalization Challenges of Large Language Models via Code Completion","date":"2024-03-12","arxiv_id":"2403.07865","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["renqibing/CodeAttack"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/moralbert-detecting-moral-values-in-social","slug":"moralbert-detecting-moral-values-in-social","title":"MoralBERT: A Fine-Tuned Language Model for Capturing Moral Values in Social Discussions","date":"2024-03-12","arxiv_id":"2403.07678","n_code_links":1,"syntology":null},{"paper":null,"slug":"rethinking-generative-large-language-model","title":"Rethinking Generative Large Language Model Evaluation for Semantic Comprehension","date":"2024-03-12","arxiv_id":"2403.07872","n_code_links":0,"syntology":null},{"paper":null,"slug":"sifid-reassess-summary-factual-inconsistency","title":"SIFiD: Reassess Summary Factual Inconsistency Detection with LLM","date":"2024-03-12","arxiv_id":"2403.07557","n_code_links":0,"syntology":null},{"paper":"/paper/stabletoolbench-towards-stable-large-scale","slug":"stabletoolbench-towards-stable-large-scale","title":"StableToolBench: Towards Stable Large-Scale Benchmarking on Tool Learning of Large Language Models","date":"2024-03-12","arxiv_id":"2403.07714","n_code_links":4,"syntology":{"ran":11,"of":15,"n_ran_checked":9,"n_instrument":2,"unverified":4,"pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["thunlp-mt/stabletoolbench","zhichengg/stabletoolbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"stress-index-strategy-enhanced-with-financial","title":"Stress index strategy enhanced with financial news sentiment analysis for the equity markets","date":"2024-03-12","arxiv_id":"2404.00012","n_code_links":0,"syntology":null},{"paper":"/paper/training-small-multimodal-models-to-bridge","slug":"training-small-multimodal-models-to-bridge","title":"Towards a clinically accessible radiology foundation model: open-access and lightweight, with automated evaluation","date":"2024-03-12","arxiv_id":"2403.08002","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":4,"n_instrument":3,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["microsoft/llava-rad"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/amharic-llama-and-llava-multimodal-llms-for","slug":"amharic-llama-and-llava-multimodal-llms-for","title":"Amharic LLaMA and LLaVA: Multimodal LLMs for Low Resource Languages","date":"2024-03-11","arxiv_id":"2403.06354","n_code_links":1,"syntology":null},{"paper":null,"slug":"guiding-clinical-reasoning-with-large","title":"Guiding Clinical Reasoning with Large Language Models via Knowledge Seeds","date":"2024-03-11","arxiv_id":"2403.06609","n_code_links":0,"syntology":null},{"paper":"/paper/smart-automatically-scaling-down-language","slug":"smart-automatically-scaling-down-language","title":"SMART: Automatically Scaling Down Language Models with Accuracy Guarantees for Reduced Processing Fees","date":"2024-03-11","arxiv_id":"2403.13835","n_code_links":1,"syntology":null},{"paper":null,"slug":"unraveling-the-mystery-of-scaling-laws-part-i","title":"Unraveling the Mystery of Scaling Laws: Part I","date":"2024-03-11","arxiv_id":"2403.06563","n_code_links":0,"syntology":null},{"paper":"/paper/autoeval-done-right-using-synthetic-data-for","slug":"autoeval-done-right-using-synthetic-data-for","title":"AutoEval Done Right: Using Synthetic Data for Model Evaluation","date":"2024-03-09","arxiv_id":"2403.07008","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pierreboyeau/autoeval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clinicalmamba-a-generative-clinical-language","slug":"clinicalmamba-a-generative-clinical-language","title":"ClinicalMamba: A Generative Clinical Language Model on Longitudinal Clinical Notes","date":"2024-03-09","arxiv_id":"2403.05795","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["whaleloops/clinicalmamba"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-benchmark-of-domain-adapted-large-language","slug":"a-benchmark-of-domain-adapted-large-language","title":"A Dataset and Benchmark for Hospital Course Summarization with Adapted Large Language Models","date":"2024-03-08","arxiv_id":"2403.05720","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-in-depth-evaluation-of-gpt-4-in-sentence","title":"An In-depth Evaluation of GPT-4 in Sentence Simplification with Error-based Human Assessment","date":"2024-03-08","arxiv_id":"2403.04963","n_code_links":0,"syntology":null},{"paper":"/paper/are-large-language-models-aligned-with-people","slug":"are-large-language-models-aligned-with-people","title":"Are Large Language Models Aligned with People's Social Intuitions for Human-Robot Interactions?","date":"2024-03-08","arxiv_id":"2403.05701","n_code_links":1,"syntology":null},{"paper":"/paper/can-t-remember-details-in-long-documents-you","slug":"can-t-remember-details-in-long-documents-you","title":"Can't Remember Details in Long Documents? You Need Some R&R","date":"2024-03-08","arxiv_id":"2403.05004","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["casetext/r-and-r"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cost-performance-optimization-for-processing","title":"Cost-Performance Optimization for Processing Low-Resource Language Tasks Using Commercial LLMs","date":"2024-03-08","arxiv_id":"2403.05434","n_code_links":0,"syntology":null},{"paper":null,"slug":"decomposing-vision-based-llm-predictions-for","title":"How Well Do Multi-modal LLMs Interpret CT Scans? An Auto-Evaluation Framework for Analyses","date":"2024-03-08","arxiv_id":"2403.05680","n_code_links":0,"syntology":null},{"paper":"/paper/erbench-an-entity-relationship-based","slug":"erbench-an-entity-relationship-based","title":"ERBench: An Entity-Relationship based Automatically Verifiable Hallucination Benchmark for Large Language Models","date":"2024-03-08","arxiv_id":"2403.05266","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dilab-kaist/erbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","slug":"gemini-1-5-unlocking-multimodal-understanding","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","date":"2024-03-08","arxiv_id":"2403.05530","n_code_links":1,"syntology":null},{"paper":"/paper/llm4decompile-decompiling-binary-code-with","slug":"llm4decompile-decompiling-binary-code-with","title":"LLM4Decompile: Decompiling Binary Code with Large Language Models","date":"2024-03-08","arxiv_id":"2403.05286","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["albertan017/LLM4Decompile"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/rat-retrieval-augmented-thoughts-elicit","slug":"rat-retrieval-augmented-thoughts-elicit","title":"RAT: Retrieval Augmented Thoughts Elicit Context-Aware Reasoning in Long-Horizon Generation","date":"2024-03-08","arxiv_id":"2403.05313","n_code_links":1,"syntology":null},{"paper":null,"slug":"will-gpt-4-run-doom","title":"Will GPT-4 Run DOOM?","date":"2024-03-08","arxiv_id":"2403.05468","n_code_links":0,"syntology":null},{"paper":null,"slug":"feedback-generation-for-programming-exercises","title":"Feedback-Generation for Programming Exercises With GPT-4","date":"2024-03-07","arxiv_id":"2403.04449","n_code_links":0,"syntology":null},{"paper":"/paper/halueval-wild-evaluating-hallucinations-of","slug":"halueval-wild-evaluating-hallucinations-of","title":"HaluEval-Wild: Evaluating Hallucinations of Language Models in the Wild","date":"2024-03-07","arxiv_id":"2403.04307","n_code_links":1,"syntology":null},{"paper":"/paper/llms-in-the-imaginarium-tool-learning-through","slug":"llms-in-the-imaginarium-tool-learning-through","title":"LLMs in the Imaginarium: Tool Learning through Simulated Trial and Error","date":"2024-03-07","arxiv_id":"2403.04746","n_code_links":1,"syntology":{"ran":6,"of":13,"n_ran_checked":6,"n_instrument":0,"unverified":7,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["microsoft/simulated-trial-and-error"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"assessing-the-aesthetic-evaluation","title":"Assessing the Aesthetic Evaluation Capabilities of GPT-4 with Vision: Insights from Group and Individual Assessments","date":"2024-03-06","arxiv_id":"2403.03594","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-do-analytical","title":"Can Large Language Models do Analytical Reasoning?","date":"2024-03-06","arxiv_id":"2403.04031","n_code_links":0,"syntology":null},{"paper":null,"slug":"designing-informative-metrics-for-few-shot","title":"Designing Informative Metrics for Few-Shot Example Selection","date":"2024-03-06","arxiv_id":"2403.03861","n_code_links":0,"syntology":null},{"paper":null,"slug":"general2specialized-llms-translation-for-e","title":"General2Specialized LLMs Translation for E-commerce","date":"2024-03-06","arxiv_id":"2403.03689","n_code_links":0,"syntology":null},{"paper":"/paper/pptc-r-benchmark-towards-evaluating-the","slug":"pptc-r-benchmark-towards-evaluating-the","title":"PPTC-R benchmark: Towards Evaluating the Robustness of Large Language Models for PowerPoint Task Completion","date":"2024-03-06","arxiv_id":"2403.03788","n_code_links":1,"syntology":null},{"paper":"/paper/rapidly-developing-high-quality-instruction","slug":"rapidly-developing-high-quality-instruction","title":"Rapidly Developing High-quality Instruction Data and Evaluation Benchmark for Large Language Models with Minimal Human Effort: A Case Study on Japanese","date":"2024-03-06","arxiv_id":"2403.03690","n_code_links":2,"syntology":null},{"paper":null,"slug":"ai-insights-a-case-study-on-utilizing-chatgpt","title":"AI Insights: A Case Study on Utilizing ChatGPT Intelligence for Research Paper Analysis","date":"2024-03-05","arxiv_id":"2403.03293","n_code_links":0,"syntology":null},{"paper":"/paper/an-empirical-study-of-llm-as-a-judge-for-llm","slug":"an-empirical-study-of-llm-as-a-judge-for-llm","title":"An Empirical Study of LLM-as-a-Judge for LLM Evaluation: Fine-tuned Judge Model is not a General Substitute for GPT-4","date":"2024-03-05","arxiv_id":"2403.02839","n_code_links":1,"syntology":{"ran":15,"of":18,"n_ran_checked":15,"n_instrument":0,"unverified":3,"pointer_only":18,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["huihuichyan/unlimitedjudge"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clevr-poc-reasoning-intensive-visual-question","title":"CLEVR-POC: Reasoning-Intensive Visual Question Answering in Partially Observable Environments","date":"2024-03-05","arxiv_id":"2403.03203","n_code_links":0,"syntology":null},{"paper":null,"slug":"emerging-synergies-between-large-language","title":"Emerging Synergies Between Large Language Models and Machine Learning in Ecommerce Recommendations","date":"2024-03-05","arxiv_id":"2403.02760","n_code_links":0,"syntology":null},{"paper":"/paper/injecagent-benchmarking-indirect-prompt","slug":"injecagent-benchmarking-indirect-prompt","title":"InjecAgent: Benchmarking Indirect Prompt Injections in Tool-Integrated Large Language Model Agents","date":"2024-03-05","arxiv_id":"2403.02691","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["uiuc-kang-lab/injecagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/paradise-evaluating-implicit-planning-skills","slug":"paradise-evaluating-implicit-planning-skills","title":"PARADISE: Evaluating Implicit Planning Skills of Language Models with Procedural Warnings and Tips Dataset","date":"2024-03-05","arxiv_id":"2403.03167","n_code_links":1,"syntology":null},{"paper":null,"slug":"scope-of-large-language-models-for-mining","title":"Scope of Large Language Models for Mining Emerging Opinions in Online Health Discourse","date":"2024-03-05","arxiv_id":"2403.03336","n_code_links":0,"syntology":null},{"paper":null,"slug":"sniffer-multimodal-large-language-model-for","title":"SNIFFER: Multimodal Large Language Model for Explainable Out-of-Context Misinformation Detection","date":"2024-03-05","arxiv_id":"2403.03170","n_code_links":0,"syntology":null}],"record_sha256":"5e4ed7151c8f09f75481439e5621b945fbb702ae9dd4aebce8df6f694d1d9ae4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}