{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt/papers/7","list_of":"/method/gpt","method":"GPT","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":7,"pages_in_order":13,"rows_per_page":100,"rows":[601,700],"of":1212,"counts":{"archive_papers_tagged":1212,"with_a_code_link":453,"where_syntology_ran_a_sample":152,"not_listed_spam_title":0,"listed":1212,"listed_where_code_ran":152,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":130,"every_run_a_failure_of_syntologys_instrument":22,"listed_with_a_run_with_no_instrument_failure":130,"listed_every_run_a_failure_of_syntologys_instrument":22,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt","prev":"/method/gpt/papers/6","next":"/method/gpt/papers/8","papers":[{"paper":null,"slug":"japanese-english-sentence-translation","title":"Japanese-English Sentence Translation Exercises Dataset for Automatic Grading","date":"2024-03-06","arxiv_id":"2403.03396","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-naive-approaches-to-tell-apart-llms","slug":"exploring-naive-approaches-to-tell-apart-llms","title":"Exploring Naive Approaches to Tell Apart LLMs Productions from Human-written Text","date":"2024-03-05","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/jmi-at-semeval-2024-task-3-two-step-approach","slug":"jmi-at-semeval-2024-task-3-two-step-approach","title":"JMI at SemEval 2024 Task 3: Two-step approach for multimodal ECAC using in-context learning with GPT and instruction-tuned Llama models","date":"2024-03-05","arxiv_id":"2403.04798","n_code_links":1,"syntology":{"ran":5,"of":12,"n_ran_checked":5,"n_instrument":0,"unverified":7,"pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["cmooncs/semeval-2024_multimodal_ecpe"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"knowledge-graphs-as-context-sources-for-llm","title":"Knowledge Graphs as Context Sources for LLM-Based Explanations of Learning Recommendations","date":"2024-03-05","arxiv_id":"2403.03008","n_code_links":0,"syntology":null},{"paper":"/paper/towards-democratized-flood-risk-management-an","slug":"towards-democratized-flood-risk-management-an","title":"Towards Democratized Flood Risk Management: An Advanced AI Assistant Enabled by GPT-4 for Enhanced Interpretability and Public Engagement","date":"2024-03-05","arxiv_id":"2403.03188","n_code_links":2,"syntology":null},{"paper":null,"slug":"automated-generation-of-multiple-choice-cloze","title":"Automated Generation of Multiple-Choice Cloze Questions for Assessing English Vocabulary Using GPT-turbo 3.5","date":"2024-03-04","arxiv_id":"2403.02078","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-llms-generate-architectural-design","title":"Can LLMs Generate Architectural Design Decisions? -An Exploratory Empirical study","date":"2024-03-04","arxiv_id":"2403.01709","n_code_links":0,"syntology":null},{"paper":"/paper/sciassess-benchmarking-llm-proficiency-in","slug":"sciassess-benchmarking-llm-proficiency-in","title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","date":"2024-03-04","arxiv_id":"2403.01976","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sci-assess/sciassess"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gender-bias-in-large-language-models-across","title":"Gender Bias in Large Language Models across Multiple Languages","date":"2024-03-01","arxiv_id":"2403.00277","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompting-chatgpt-for-translation-a","title":"Prompting ChatGPT for Translation: A Comparative Analysis of Translation Brief and Persona Prompts","date":"2024-02-29","arxiv_id":"2403.00127","n_code_links":0,"syntology":null},{"paper":null,"slug":"rl-gpt-integrating-reinforcement-learning-and","title":"RL-GPT: Integrating Reinforcement Learning and Code-as-policy","date":"2024-02-29","arxiv_id":"2402.19299","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-gpt-improve-the-state-of-prior","title":"Can GPT Improve the State of Prior Authorization via Guideline Based Automated Question Answering?","date":"2024-02-28","arxiv_id":"2402.18419","n_code_links":0,"syntology":null},{"paper":"/paper/clustering-and-ranking-diversity-preserved","slug":"clustering-and-ranking-diversity-preserved","title":"Clustering and Ranking: Diversity-preserved Instruction Selection through Expert-aligned Quality Estimation","date":"2024-02-28","arxiv_id":"2402.18191","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ironbeliever/car"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/gradient-free-adaptive-global-pruning-for-pre","slug":"gradient-free-adaptive-global-pruning-for-pre","title":"SparseLLM: Towards Global Pruning for Pre-trained Language Models","date":"2024-02-28","arxiv_id":"2402.17946","n_code_links":2,"syntology":{"ran":3,"of":9,"n_ran_checked":2,"n_instrument":1,"unverified":6,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["baithebest/adagp","baithebest/sparsellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-language-model-based-framework-for-new","slug":"a-language-model-based-framework-for-new","title":"A Language Model based Framework for New Concept Placement in Ontologies","date":"2024-02-27","arxiv_id":"2402.17897","n_code_links":1,"syntology":null},{"paper":null,"slug":"cocoa-cbt-based-conversational-counseling","title":"COCOA: CBT-based Conversational Counseling Agent using Memory Specialized in Cognitive Distortions and Dynamic Prompt","date":"2024-02-27","arxiv_id":"2402.17546","n_code_links":0,"syntology":null},{"paper":null,"slug":"prp-propagating-universal-perturbations-to","title":"PRP: Propagating Universal Perturbations to Attack Large Language Model Guard-Rails","date":"2024-02-24","arxiv_id":"2402.15911","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-first-look-at-gpt-apps-landscape-and","title":"A First Look at GPT Apps: Landscape and Vulnerability","date":"2024-02-23","arxiv_id":"2402.15105","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-understanding-counseling","title":"Towards Understanding Counseling Conversations: Domain Knowledge and Large Language Models","date":"2024-02-22","arxiv_id":"2402.14200","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-evaluation-of-large-language-models-in","title":"An Evaluation of Large Language Models in Bioinformatics Research","date":"2024-02-21","arxiv_id":"2402.13714","n_code_links":0,"syntology":null},{"paper":null,"slug":"hallucinations-or-attention-misdirection-the","title":"Hallucinations or Attention Misdirection? The Path to Strategic Value Extraction in Business Using Large Language Models","date":"2024-02-21","arxiv_id":"2402.14002","n_code_links":0,"syntology":null},{"paper":"/paper/synfac-edit-synthetic-imitation-edit-feedback","slug":"synfac-edit-synthetic-imitation-edit-feedback","title":"SYNFAC-EDIT: Synthetic Imitation Edit Feedback for Factual Alignment in Clinical Summarization","date":"2024-02-21","arxiv_id":"2402.13919","n_code_links":1,"syntology":null},{"paper":"/paper/chatel-entity-linking-with-chatbots","slug":"chatel-entity-linking-with-chatbots","title":"ChatEL: Entity Linking with Chatbots","date":"2024-02-20","arxiv_id":"2402.14858","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-the-system-message-really-important-to","title":"Is the System Message Really Important to Jailbreaks in Large Language Models?","date":"2024-02-20","arxiv_id":"2402.14857","n_code_links":0,"syntology":null},{"paper":"/paper/analobench-benchmarking-the-identification-of","slug":"analobench-benchmarking-the-identification-of","title":"AnaloBench: Benchmarking the Identification of Abstract and Long-context Analogies","date":"2024-02-19","arxiv_id":"2402.12370","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jhu-clsp/analogical-reasoning","JHU-CLSP/AnaloBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"inference-to-the-best-explanation-in-large","title":"Inference to the Best Explanation in Large Language Models","date":"2024-02-16","arxiv_id":"2402.10767","n_code_links":0,"syntology":null},{"paper":"/paper/network-formation-and-dynamics-among-multi","slug":"network-formation-and-dynamics-among-multi","title":"Network Formation and Dynamics Among Multi-LLMs","date":"2024-02-16","arxiv_id":"2402.10659","n_code_links":1,"syntology":null},{"paper":null,"slug":"best-arm-identification-for-prompt-learning","title":"Efficient Prompt Optimization Through the Lens of Best Arm Identification","date":"2024-02-15","arxiv_id":"2402.09723","n_code_links":0,"syntology":null},{"paper":null,"slug":"using-counterfactual-tasks-to-evaluate-the","title":"Using Counterfactual Tasks to Evaluate the Generality of Analogical Reasoning in Large Language Models","date":"2024-02-14","arxiv_id":"2402.08955","n_code_links":0,"syntology":null},{"paper":null,"slug":"eliciting-big-five-personality-traits-in","title":"Eliciting Personality Traits in Large Language Models","date":"2024-02-13","arxiv_id":"2402.08341","n_code_links":0,"syntology":null},{"paper":null,"slug":"sentinels-of-the-stream-unleashing-large","title":"Sentinels of the Stream: Unleashing Large Language Models for Dynamic Packet Classification in Software Defined Networks -- Position Paper","date":"2024-02-10","arxiv_id":"2402.07950","n_code_links":0,"syntology":null},{"paper":"/paper/entgpt-linking-generative-large-language","slug":"entgpt-linking-generative-large-language","title":"EntGPT: Linking Generative Large Language Models with Knowledge Bases","date":"2024-02-09","arxiv_id":"2402.06738","n_code_links":1,"syntology":null},{"paper":null,"slug":"learn-to-be-efficient-build-structured","title":"Learn To be Efficient: Build Structured Sparsity in Large Language Models","date":"2024-02-09","arxiv_id":"2402.06126","n_code_links":0,"syntology":null},{"paper":null,"slug":"named-entity-recognition-for-address","title":"Named Entity Recognition for Address Extraction in Speech-to-Text Transcriptions Using Synthetic Data","date":"2024-02-08","arxiv_id":"2402.05545","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-legal-reasoning-the-integration-of","title":"Advancing Legal Reasoning: The Integration of AI to Navigate Complexities and Biases in Global Jurisprudence with Semi-Automated Arbitration Processes (SAAPs)","date":"2024-02-06","arxiv_id":"2402.04140","n_code_links":0,"syntology":null},{"paper":null,"slug":"cehr-gpt-generating-electronic-health-records","title":"CEHR-GPT: Generating Electronic Health Records with Chronological Patient Timelines","date":"2024-02-06","arxiv_id":"2402.04400","n_code_links":0,"syntology":null},{"paper":"/paper/pard-permutation-invariant-autoregressive","slug":"pard-permutation-invariant-autoregressive","title":"Pard: Permutation-Invariant Autoregressive Diffusion for Graph Generation","date":"2024-02-06","arxiv_id":"2402.03687","n_code_links":1,"syntology":null},{"paper":"/paper/conversation-reconstruction-attack-against","slug":"conversation-reconstruction-attack-against","title":"Reconstruct Your Previous Conversations! Comprehensively Investigating Privacy Leakage Risks in Conversations with GPT Models","date":"2024-02-05","arxiv_id":"2402.02987","n_code_links":1,"syntology":null},{"paper":"/paper/a-graph-is-worth-k-words-euclideanizing-graph","slug":"a-graph-is-worth-k-words-euclideanizing-graph","title":"A Graph is Worth $K$ Words: Euclideanizing Graph using Pure Transformer","date":"2024-02-04","arxiv_id":"2402.02464","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["A4Bio/GraphsGPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/autotimes-autoregressive-time-series","slug":"autotimes-autoregressive-time-series","title":"AutoTimes: Autoregressive Time Series Forecasters via Large Language Models","date":"2024-02-04","arxiv_id":"2402.02370","n_code_links":1,"syntology":null},{"paper":"/paper/tsis-a-supplementary-algorithm-to-t-smiles","slug":"tsis-a-supplementary-algorithm-to-t-smiles","title":"Hierarchical Structure Enhances the Convergence and Generalizability of Linear Molecular Representation","date":"2024-02-03","arxiv_id":"2402.02164","n_code_links":1,"syntology":null},{"paper":null,"slug":"comet-generating-commit-messages-using-delta","title":"COMET: Generating Commit Messages using Delta Graph Context Representation","date":"2024-02-02","arxiv_id":"2402.01841","n_code_links":0,"syntology":null},{"paper":"/paper/improving-sequential-recommendations-with","slug":"improving-sequential-recommendations-with","title":"Improving Sequential Recommendations with LLMs","date":"2024-02-02","arxiv_id":"2402.01339","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dh-r/llm-sequential-recommendation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sparql-generation-with-entity-pre-trained-gpt","slug":"sparql-generation-with-entity-pre-trained-gpt","title":"SPARQL Generation with Entity Pre-trained GPT for KG Question Answering","date":"2024-02-01","arxiv_id":"2402.00969","n_code_links":1,"syntology":null},{"paper":null,"slug":"global-liar-factuality-of-llms-over-time-and","title":"Global-Liar: Factuality of LLMs over Time and Geographic Regions","date":"2024-01-31","arxiv_id":"2401.17839","n_code_links":0,"syntology":null},{"paper":null,"slug":"real-sparks-of-artificial-intelligence-and","title":"Real Sparks of Artificial Intelligence and the Importance of Inner Interpretability","date":"2024-01-31","arxiv_id":"2402.00901","n_code_links":0,"syntology":null},{"paper":null,"slug":"diverse-but-divisive-llms-can-exaggerate","title":"Diverse, but Divisive: LLMs Can Exaggerate Gender Differences in Opinion Related to Harms of Misinformation","date":"2024-01-29","arxiv_id":"2401.16558","n_code_links":0,"syntology":null},{"paper":"/paper/e-eval-a-comprehensive-chinese-k-12-education","slug":"e-eval-a-comprehensive-chinese-k-12-education","title":"E-EVAL: A Comprehensive Chinese K-12 Education Evaluation Benchmark for Large Language Models","date":"2024-01-29","arxiv_id":"2401.15927","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ai-edu-lab/e-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"trackgpt-a-generative-pre-trained-transformer","title":"TrackGPT -- A generative pre-trained transformer for cross-domain entity trajectory forecasting","date":"2024-01-29","arxiv_id":"2402.00066","n_code_links":0,"syntology":null},{"paper":"/paper/convosense-overcoming-monotonous-commonsense","slug":"convosense-overcoming-monotonous-commonsense","title":"ConvoSense: Overcoming Monotonous Commonsense Inferences for Conversational AI","date":"2024-01-27","arxiv_id":"2401.15471","n_code_links":1,"syntology":null},{"paper":null,"slug":"geodecoder-empowering-multimodal-map","title":"GeoDecoder: Empowering Multimodal Map Understanding","date":"2024-01-26","arxiv_id":"2401.15118","n_code_links":0,"syntology":null},{"paper":"/paper/chat-gpt-v-bert-dawn-of-justice-for-semantic","slug":"chat-gpt-v-bert-dawn-of-justice-for-semantic","title":"(Chat)GPT v BERT: Dawn of Justice for Semantic Change Detection","date":"2024-01-25","arxiv_id":"2401.14040","n_code_links":1,"syntology":null},{"paper":null,"slug":"automated-root-causing-of-cloud-incidents","title":"Automated Root Causing of Cloud Incidents using In-Context Learning with GPT-4","date":"2024-01-24","arxiv_id":"2401.13810","n_code_links":0,"syntology":null},{"paper":null,"slug":"discovering-mathematical-formulas-from-data","title":"Discovering Mathematical Formulas from Data via GPT-guided Monte Carlo Tree Search","date":"2024-01-24","arxiv_id":"2401.14424","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluation-of-general-large-language-models","title":"Evaluation of General Large Language Models in Contextually Assessing Semantic Concepts Extracted from Adult Critical Care Electronic Health Record Notes","date":"2024-01-24","arxiv_id":"2401.13588","n_code_links":0,"syntology":null},{"paper":null,"slug":"graph-guided-question-answer-generation-for","title":"Graph Guided Question Answer Generation for Procedural Question-Answering","date":"2024-01-24","arxiv_id":"2401.13594","n_code_links":0,"syntology":null},{"paper":"/paper/how-good-is-chatgpt-at-face-biometrics-a","slug":"how-good-is-chatgpt-at-face-biometrics-a","title":"How Good is ChatGPT at Face Biometrics? A First Look into Recognition, Soft Biometrics, and Explainability","date":"2024-01-24","arxiv_id":"2401.13641","n_code_links":1,"syntology":null},{"paper":null,"slug":"segment-any-cell-a-sam-based-auto-prompting","title":"Segment Any Cell: A SAM-based Auto-prompting Fine-tuning Framework for Nuclei Segmentation","date":"2024-01-24","arxiv_id":"2401.13220","n_code_links":0,"syntology":null},{"paper":"/paper/trove-inducing-verifiable-and-efficient","slug":"trove-inducing-verifiable-and-efficient","title":"TroVE: Inducing Verifiable and Efficient Toolboxes for Solving Programmatic Tasks","date":"2024-01-23","arxiv_id":"2401.12869","n_code_links":1,"syntology":{"ran":16,"of":19,"n_ran_checked":16,"n_instrument":0,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zorazrw/trove"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-in-context-learning-via-linear","slug":"enhancing-in-context-learning-via-linear","title":"Enhancing In-context Learning via Linear Probe Calibration","date":"2024-01-22","arxiv_id":"2401.12406","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mominabbass/linc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/chex-gpt-harnessing-large-language-models-for","slug":"chex-gpt-harnessing-large-language-models-for","title":"CheX-GPT: Harnessing Large Language Models for Enhanced Chest X-ray Report Labeling","date":"2024-01-21","arxiv_id":"2401.11505","n_code_links":2,"syntology":null},{"paper":null,"slug":"enhancing-recommendation-diversity-by-re","title":"Enhancing Recommendation Diversity by Re-ranking with Large Language Models","date":"2024-01-21","arxiv_id":"2401.11506","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforcement-learning-for-question-answering","title":"Reinforcement learning for question answering in programming domain using public community scoring as a human feedback","date":"2024-01-19","arxiv_id":"2401.10882","n_code_links":0,"syntology":null},{"paper":"/paper/chatqa-building-gpt-4-level-conversational-qa","slug":"chatqa-building-gpt-4-level-conversational-qa","title":"ChatQA: Surpassing GPT-4 on Conversational QA and RAG","date":"2024-01-18","arxiv_id":"2401.10225","n_code_links":0,"syntology":null},{"paper":"/paper/code-prompting-elicits-conditional-reasoning","slug":"code-prompting-elicits-conditional-reasoning","title":"Code Prompting Elicits Conditional Reasoning Abilities in Text+Code LLMs","date":"2024-01-18","arxiv_id":"2401.10065","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ukplab/arxiv2024-conditional-reasoning-llms"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"image-translation-as-diffusion-visual","title":"Image Translation as Diffusion Visual Programmers","date":"2024-01-18","arxiv_id":"2401.09742","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-robustness-of-llm-synthetic-text","title":"Enhancing Robustness of LLM-Synthetic Text Detectors for Academic Writing: A Comprehensive Analysis","date":"2024-01-16","arxiv_id":"2401.08046","n_code_links":0,"syntology":null},{"paper":"/paper/exploiting-inter-layer-expert-affinity-for","slug":"exploiting-inter-layer-expert-affinity-for","title":"Exploiting Inter-Layer Expert Affinity for Accelerating Mixture-of-Experts Model Inference","date":"2024-01-16","arxiv_id":"2401.08383","n_code_links":1,"syntology":null},{"paper":"/paper/rotbench-a-multi-level-benchmark-for","slug":"rotbench-a-multi-level-benchmark-for","title":"RoTBench: A Multi-Level Benchmark for Evaluating the Robustness of Large Language Models in Tool Learning","date":"2024-01-16","arxiv_id":"2401.08326","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["junjie-ye/rotbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-novel-approach-for-automatic-program-repair","slug":"a-novel-approach-for-automatic-program-repair","title":"A Novel Approach for Automatic Program Repair using Round-Trip Translation with Large Language Models","date":"2024-01-15","arxiv_id":"2401.07994","n_code_links":1,"syntology":null},{"paper":"/paper/harnessing-large-language-models-over","slug":"harnessing-large-language-models-over","title":"Harnessing Large Language Models Over Transformer Models for Detecting Bengali Depressive Social Media Text: A Comprehensive Study","date":"2024-01-14","arxiv_id":"2401.07310","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-be-homo-economicus-can-an-llm","title":"Learning to be Homo Economicus: Can an LLM Learn Preferences from Choice","date":"2024-01-14","arxiv_id":"2401.07345","n_code_links":0,"syntology":null},{"paper":null,"slug":"mapgpt-map-guided-prompting-for-unified","title":"MapGPT: Map-Guided Prompting with Adaptive Path Planning for Vision-and-Language Navigation","date":"2024-01-14","arxiv_id":"2401.07314","n_code_links":0,"syntology":null},{"paper":null,"slug":"streamlining-the-selection-phase-of","title":"Streamlining the Selection Phase of Systematic Literature Reviews (SLRs) Using AI-Enabled GPT-4 Assistant API","date":"2024-01-14","arxiv_id":"2402.18582","n_code_links":0,"syntology":null},{"paper":"/paper/a-novel-multi-stage-prompting-approach-for","slug":"a-novel-multi-stage-prompting-approach-for","title":"A Novel Multi-Stage Prompting Approach for Language Agnostic MCQ Generation using GPT","date":"2024-01-13","arxiv_id":"2401.07098","n_code_links":1,"syntology":null},{"paper":null,"slug":"assessing-large-language-models-in-mechanical","title":"Assessing Large Language Models in Mechanical Engineering Education: A Study on Mechanics-Focused Conceptual Understanding","date":"2024-01-13","arxiv_id":"2401.12983","n_code_links":0,"syntology":null},{"paper":null,"slug":"combining-confidence-elicitation-and-sample","title":"Combining Confidence Elicitation and Sample-based Methods for Uncertainty Quantification in Misinformation Mitigation","date":"2024-01-13","arxiv_id":"2401.08694","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-based-mental-health-screening-from","title":"Prompt-based mental health screening from social media text","date":"2024-01-11","arxiv_id":"2401.05912","n_code_links":0,"syntology":null},{"paper":"/paper/i-am-a-strange-dataset-metalinguistic-tests","slug":"i-am-a-strange-dataset-metalinguistic-tests","title":"I am a Strange Dataset: Metalinguistic Tests for Language Models","date":"2024-01-10","arxiv_id":"2401.05300","n_code_links":1,"syntology":null},{"paper":null,"slug":"fighting-fire-with-fire-adversarial-prompting","title":"Fighting Fire with Fire: Adversarial Prompting to Generate a Misinformation Detection Dataset","date":"2024-01-09","arxiv_id":"2401.04481","n_code_links":0,"syntology":null},{"paper":"/paper/advancing-spatial-reasoning-in-large-language","slug":"advancing-spatial-reasoning-in-large-language","title":"Advancing Spatial Reasoning in Large Language Models: An In-Depth Evaluation and Enhancement Using the StepGame Benchmark","date":"2024-01-08","arxiv_id":"2401.03991","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-in-bioinformatics","title":"Advancing bioinformatics with large language models: components, applications and perspectives","date":"2024-01-08","arxiv_id":"2401.04155","n_code_links":0,"syntology":null},{"paper":"/paper/comparative-analysis-of-llama-and-chatgpt","slug":"comparative-analysis-of-llama-and-chatgpt","title":"Can Large Language Models Understand Molecules?","date":"2024-01-05","arxiv_id":"2402.00024","n_code_links":2,"syntology":null},{"paper":"/paper/text2mdt-extracting-medical-decision-trees","slug":"text2mdt-extracting-medical-decision-trees","title":"Text2MDT: Extracting Medical Decision Trees from Medical Texts","date":"2024-01-04","arxiv_id":"2401.02034","n_code_links":1,"syntology":null},{"paper":null,"slug":"iot-in-the-era-of-generative-ai-vision-and","title":"The Internet of Things in the Era of Generative AI: Vision and Challenges","date":"2024-01-03","arxiv_id":"2401.01923","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-zero-shot-abstractive","slug":"revisiting-zero-shot-abstractive","title":"Revisiting Zero-Shot Abstractive Summarization in the Era of Large Language Models from the Perspective of Position Bias","date":"2024-01-03","arxiv_id":"2401.01989","n_code_links":1,"syntology":null},{"paper":"/paper/a-computational-framework-for-behavioral","slug":"a-computational-framework-for-behavioral","title":"A Computational Framework for Behavioral Assessment of LLM Therapists","date":"2024-01-01","arxiv_id":"2401.00820","n_code_links":1,"syntology":null},{"paper":"/paper/seed-bench-benchmarking-multimodal-large","slug":"seed-bench-benchmarking-multimodal-large","title":"SEED-Bench: Benchmarking Multimodal Large Language Models","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"trace-and-edit-relation-associations-in-gpt","title":"Trace and Edit Relation Associations in GPT","date":"2023-12-30","arxiv_id":"2401.02976","n_code_links":0,"syntology":null},{"paper":"/paper/gemini-in-reasoning-unveiling-commonsense-in","slug":"gemini-in-reasoning-unveiling-commonsense-in","title":"Gemini in Reasoning: Unveiling Commonsense in Multimodal Large Language Models","date":"2023-12-29","arxiv_id":"2312.17661","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-model-for-causal-decision","title":"LLM4Causal: Democratized Causal Tools for Everyone via Large Language Model","date":"2023-12-28","arxiv_id":"2312.17122","n_code_links":0,"syntology":null},{"paper":null,"slug":"pangu-p-enhancing-language-model","title":"PanGu-$π$: Enhancing Language Model Architectures via Nonlinearity Compensation","date":"2023-12-27","arxiv_id":"2312.17276","n_code_links":0,"syntology":null},{"paper":null,"slug":"chartbench-a-benchmark-for-complex-visual","title":"ChartBench: A Benchmark for Complex Visual Reasoning in Charts","date":"2023-12-26","arxiv_id":"2312.15915","n_code_links":0,"syntology":null},{"paper":null,"slug":"do-llm-agents-exhibit-social-behavior","title":"Do LLM Agents Exhibit Social Behavior?","date":"2023-12-23","arxiv_id":"2312.15198","n_code_links":0,"syntology":null},{"paper":"/paper/understanding-the-potential-of-fpga-based","slug":"understanding-the-potential-of-fpga-based","title":"Understanding the Potential of FPGA-Based Spatial Acceleration for Large Language Model Inference","date":"2023-12-23","arxiv_id":"2312.15159","n_code_links":1,"syntology":null},{"paper":"/paper/chatgpt-as-a-commenter-to-the-news-can-llms","slug":"chatgpt-as-a-commenter-to-the-news-can-llms","title":"ChatGPT as a commenter to the news: can LLMs generate human-like opinions?","date":"2023-12-21","arxiv_id":"2312.13961","n_code_links":1,"syntology":null},{"paper":"/paper/de-novo-drug-design-using-reinforcement-1","slug":"de-novo-drug-design-using-reinforcement-1","title":"De novo Drug Design using Reinforcement Learning with Multiple GPT Agents","date":"2023-12-21","arxiv_id":"2401.06155","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hxyfighter/molrl-mgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/provfl-client-driven-interpretability-of","slug":"provfl-client-driven-interpretability-of","title":"TraceFL: Interpretability-Driven Debugging in Federated Learning via Neuron Provenance","date":"2023-12-21","arxiv_id":"2312.13632","n_code_links":2,"syntology":null},{"paper":"/paper/domain-specific-code-language-models","slug":"domain-specific-code-language-models","title":"MonoCoder: Domain-Specific Code Language Model for HPC Codes and Tasks","date":"2023-12-20","arxiv_id":"2312.13322","n_code_links":3,"syntology":null},{"paper":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"2a0a73e1af5644c8643ab1a37fbb0aaed9cf68251a3ecb3809231ba18c1f6131","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}