{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/13","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":13,"pages_in_order":29,"rows_per_page":100,"rows":[1201,1300],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/12","next":"/method/gpt-4/papers/14","papers":[{"paper":"/paper/patient-ps-using-large-language-models-to","slug":"patient-ps-using-large-language-models-to","title":"PATIENT-Ψ: Using Large Language Models to Simulate Patients for Training Mental Health Professionals","date":"2024-05-30","arxiv_id":"2405.19660","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ruiyiw/patient-psi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/perteval-unveiling-real-knowledge-capacity-of","slug":"perteval-unveiling-real-knowledge-capacity-of","title":"PertEval: Unveiling Real Knowledge Capacity of LLMs with Knowledge-Invariant Perturbations","date":"2024-05-30","arxiv_id":"2405.19740","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aigc-apps/perteval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"phantom-general-trigger-attacks-on-retrieval","title":"Phantom: General Trigger Attacks on Retrieval Augmented Language Generation","date":"2024-05-30","arxiv_id":"2405.20485","n_code_links":0,"syntology":null},{"paper":"/paper/preference-alignment-with-flow-matching","slug":"preference-alignment-with-flow-matching","title":"Preference Alignment with Flow Matching","date":"2024-05-30","arxiv_id":"2405.19806","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jadehaus/preference-flow-matching"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"are-you-sure-rank-them-again-repeated-ranking","title":"Are You Sure? Rank Them Again: Repeated Ranking For Better Preference Datasets","date":"2024-05-29","arxiv_id":"2405.18952","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-agreement-diagnosing-the-rationale","slug":"beyond-agreement-diagnosing-the-rationale","title":"Beyond Agreement: Diagnosing the Rationale Alignment of Automated Essay Scoring Methods based on Linguistically-informed Counterfactuals","date":"2024-05-29","arxiv_id":"2405.19433","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-based-hierarchical-concept-decomposition","title":"LLM-based Hierarchical Concept Decomposition for Interpretable Fine-Grained Image Classification","date":"2024-05-29","arxiv_id":"2405.18672","n_code_links":0,"syntology":null},{"paper":"/paper/llms-achieve-adult-human-performance-on","slug":"llms-achieve-adult-human-performance-on","title":"LLMs achieve adult human performance on higher-order theory of mind tasks","date":"2024-05-29","arxiv_id":"2405.18870","n_code_links":0,"syntology":null},{"paper":"/paper/pediatricsgpt-large-language-models-as","slug":"pediatricsgpt-large-language-models-as","title":"PediatricsGPT: Large Language Models as Chinese Medical Assistants for Pediatric Applications","date":"2024-05-29","arxiv_id":"2405.19266","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ydk122024/pediatricsgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/reverse-image-retrieval-cues-parametric","slug":"reverse-image-retrieval-cues-parametric","title":"Reverse Image Retrieval Cues Parametric Memory in Multimodal LLMs","date":"2024-05-29","arxiv_id":"2405.18740","n_code_links":1,"syntology":null},{"paper":null,"slug":"two-layer-retrieval-augmented-generation","title":"Two-Layer Retrieval-Augmented Generation Framework for Low-Resource Medical Question Answering Using Reddit Data: Proof-of-Concept Study","date":"2024-05-29","arxiv_id":"2405.19519","n_code_links":0,"syntology":null},{"paper":"/paper/aligning-to-thousands-of-preferences-via","slug":"aligning-to-thousands-of-preferences-via","title":"Aligning to Thousands of Preferences via System Message Generalization","date":"2024-05-28","arxiv_id":"2405.17977","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kaistAI/Janus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/an-empirical-analysis-on-large-language","slug":"an-empirical-analysis-on-large-language","title":"An Empirical Analysis on Large Language Models in Debate Evaluation","date":"2024-05-28","arxiv_id":"2406.00050","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xinyiliu0227/llm_debate_bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"benchmark-underestimates-the-readiness-of","title":"Benchmarks Underestimate the Readiness of Multi-lingual Dialogue Agents","date":"2024-05-28","arxiv_id":"2405.17840","n_code_links":0,"syntology":null},{"paper":null,"slug":"edinburgh-clinical-nlp-at-mediqa-corr-2024","title":"Edinburgh Clinical NLP at MEDIQA-CORR 2024: Guiding Large Language Models with Hints","date":"2024-05-28","arxiv_id":"2405.18028","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-objective-representation-for-numbers-in","title":"Multi-objective Representation for Numbers in Clinical Narratives: A CamemBERT-Bio-Based Alternative to Large-Scale LLMs","date":"2024-05-28","arxiv_id":"2405.18448","n_code_links":0,"syntology":null},{"paper":null,"slug":"notes-on-applicability-of-gpt-4-to-document","title":"Notes on Applicability of GPT-4 to Document Understanding","date":"2024-05-28","arxiv_id":"2405.18433","n_code_links":0,"syntology":null},{"paper":"/paper/orlm-training-large-language-models-for","slug":"orlm-training-large-language-models-for","title":"ORLM: A Customizable Framework in Training Large Models for Automated Optimization Modeling","date":"2024-05-28","arxiv_id":"2405.17743","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cardinal-operations/orlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/peering-into-the-mind-of-language-models-an","slug":"peering-into-the-mind-of-language-models-an","title":"Peering into the Mind of Language Models: An Approach for Attribution in Contextual Question Answering","date":"2024-05-28","arxiv_id":"2405.17980","n_code_links":1,"syntology":null},{"paper":null,"slug":"proof-of-quality-a-costless-paradigm-for","title":"Proof of Quality: A Costless Paradigm for Trustless Generative AI Model Inference on Blockchains","date":"2024-05-28","arxiv_id":"2405.17934","n_code_links":0,"syntology":null},{"paper":null,"slug":"realitysummary-on-demand-mixed-reality","title":"RealitySummary: Exploring On-Demand Mixed Reality Text Summarization and Question Answering using Large Language Models","date":"2024-05-28","arxiv_id":"2405.18620","n_code_links":0,"syntology":null},{"paper":"/paper/thai-winograd-schemas-a-benchmark-for-thai","slug":"thai-winograd-schemas-a-benchmark-for-thai","title":"Thai Winograd Schemas: A Benchmark for Thai Commonsense Reasoning","date":"2024-05-28","arxiv_id":"2405.18375","n_code_links":1,"syntology":null},{"paper":null,"slug":"the-battle-of-llms-a-comparative-study-in","title":"The Battle of LLMs: A Comparative Study in Conversational QA Tasks","date":"2024-05-28","arxiv_id":"2405.18344","n_code_links":0,"syntology":null},{"paper":"/paper/autoformalizing-euclidean-geometry","slug":"autoformalizing-euclidean-geometry","title":"Autoformalizing Euclidean Geometry","date":"2024-05-27","arxiv_id":"2405.17216","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":4,"n_instrument":5,"unverified":4,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","official":{"repos":["loganrjmurphy/leaneuclid"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/chess-contextual-harnessing-for-efficient-sql","slug":"chess-contextual-harnessing-for-efficient-sql","title":"CHESS: Contextual Harnessing for Efficient SQL Synthesis","date":"2024-05-27","arxiv_id":"2405.16755","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shayantalaei/chess"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"cost-efficient-knowledge-based-question","title":"Cost-efficient Knowledge-based Question Answering with Large Language Models","date":"2024-05-27","arxiv_id":"2405.17337","n_code_links":0,"syntology":null},{"paper":"/paper/generation-and-human-expert-evaluation-of","slug":"generation-and-human-expert-evaluation-of","title":"Interesting Scientific Idea Generation using Knowledge Graphs and LLMs: Evaluations with 100 Research Group Leaders","date":"2024-05-27","arxiv_id":"2405.17044","n_code_links":1,"syntology":null},{"paper":null,"slug":"llm-based-cooperative-agents-using","title":"REVECA: Adaptive Planning and Trajectory-based Validation in Cooperative Language Agents using Information Relevance and Relative Proximity","date":"2024-05-27","arxiv_id":"2405.16751","n_code_links":0,"syntology":null},{"paper":"/paper/motionllm-multimodal-motion-language-learning","slug":"motionllm-multimodal-motion-language-learning","title":"Motion-Agent: A Conversational Framework for Human Motion Generation with LLMs","date":"2024-05-27","arxiv_id":"2405.17013","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["szqwu/Motion-Agent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/reflectioncoder-learning-from-reflection","slug":"reflectioncoder-learning-from-reflection","title":"ReflectionCoder: Learning from Reflection Sequence for Enhanced One-off Code Generation","date":"2024-05-27","arxiv_id":"2405.17057","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["sensellm/reflectioncoder"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/rtl-repo-a-benchmark-for-evaluating-llms-on","slug":"rtl-repo-a-benchmark-for-evaluating-llms-on","title":"RTL-Repo: A Benchmark for Evaluating LLMs on Large-Scale RTL Design Projects","date":"2024-05-27","arxiv_id":"2405.17378","n_code_links":1,"syntology":null},{"paper":"/paper/safe-lora-the-silver-lining-of-reducing","slug":"safe-lora-the-silver-lining-of-reducing","title":"Safe LoRA: the Silver Lining of Reducing Safety Risks when Fine-tuning Large Language Models","date":"2024-05-27","arxiv_id":"2405.16833","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["ibm/safelora"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/thread-thinking-deeper-with-recursive","slug":"thread-thinking-deeper-with-recursive","title":"THREAD: Thinking Deeper with Recursive Spawning","date":"2024-05-27","arxiv_id":"2405.17402","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["philipmit/thread"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/darijabanking-a-new-resource-for-overcoming","slug":"darijabanking-a-new-resource-for-overcoming","title":"DarijaBanking: A New Resource for Overcoming Language Barriers in Banking Intent Detection for Moroccan Arabic Speakers","date":"2024-05-26","arxiv_id":"2405.16482","n_code_links":1,"syntology":null},{"paper":"/paper/meta-task-planning-for-language-agents","slug":"meta-task-planning-for-language-agents","title":"Planning with Multi-Constraints via Collaborative Language Agents","date":"2024-05-26","arxiv_id":"2405.16510","n_code_links":1,"syntology":null},{"paper":null,"slug":"comparative-analysis-of-open-source-language","title":"Comparative Analysis of Open-Source Language Models in Summarizing Medical Text Data","date":"2024-05-25","arxiv_id":"2405.16295","n_code_links":0,"syntology":null},{"paper":"/paper/confidence-under-the-hood-an-investigation","slug":"confidence-under-the-hood-an-investigation","title":"Confidence Under the Hood: An Investigation into the Confidence-Probability Alignment in Large Language Models","date":"2024-05-25","arxiv_id":"2405.16282","n_code_links":1,"syntology":{"ran":1,"of":6,"n_ran_checked":1,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["akkeshav/confidence_probability_alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"geneagent-self-verification-language-agent","title":"GeneAgent: Self-verification Language Agent for Gene Set Knowledge Discovery using Domain Databases","date":"2024-05-25","arxiv_id":"2405.16205","n_code_links":0,"syntology":null},{"paper":null,"slug":"hethub-a-heterogeneous-distributed-hybrid","title":"HETHUB: A Distributed Training System with Heterogeneous Cluster for Large-Scale Models","date":"2024-05-25","arxiv_id":"2405.16256","n_code_links":0,"syntology":null},{"paper":null,"slug":"picturing-ambiguity-a-visual-twist-on-the","title":"Picturing Ambiguity: A Visual Twist on the Winograd Schema Challenge","date":"2024-05-25","arxiv_id":"2405.16277","n_code_links":0,"syntology":null},{"paper":"/paper/stride-a-tool-assisted-llm-agent-framework","slug":"stride-a-tool-assisted-llm-agent-framework","title":"STRIDE: A Tool-Assisted LLM Agent Framework for Strategic and Interactive Decision-Making","date":"2024-05-25","arxiv_id":"2405.16376","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cyrilli/stride"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-evaluation-of-estimative-uncertainty-in","title":"An Evaluation of Estimative Uncertainty in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15185","n_code_links":0,"syntology":null},{"paper":"/paper/before-generation-align-it-a-novel-and","slug":"before-generation-align-it-a-novel-and","title":"Before Generation, Align it! A Novel and Effective Strategy for Mitigating Hallucinations in Text-to-SQL Generation","date":"2024-05-24","arxiv_id":"2405.15307","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":7,"n_instrument":1,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["quge2023/TA-SQL"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/continuously-learning-adapting-and-improving","slug":"continuously-learning-adapting-and-improving","title":"Continuously Learning, Adapting, and Improving: A Dual-Process Approach to Autonomous Driving","date":"2024-05-24","arxiv_id":"2405.15324","n_code_links":1,"syntology":null},{"paper":"/paper/culturepark-boosting-cross-cultural","slug":"culturepark-boosting-cross-cultural","title":"CulturePark: Boosting Cross-cultural Understanding in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15145","n_code_links":1,"syntology":{"ran":0,"of":7,"n_ran_checked":0,"n_instrument":0,"unverified":7,"pointer_only":7,"phrase":"0 ran · 7 unverified","official":{"repos":["scarelette/culturepark"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":[]}}},{"paper":"/paper/large-language-models-reflect-human-citation","slug":"large-language-models-reflect-human-citation","title":"Large Language Models Reflect Human Citation Patterns with a Heightened Citation Bias","date":"2024-05-24","arxiv_id":"2405.15739","n_code_links":1,"syntology":null},{"paper":null,"slug":"zero-shot-spam-email-classification-using-pre","title":"Zero-Shot Spam Email Classification Using Pre-trained Large Language Models","date":"2024-05-24","arxiv_id":"2405.15936","n_code_links":0,"syntology":null},{"paper":"/paper/a-declarative-system-for-optimizing-ai","slug":"a-declarative-system-for-optimizing-ai","title":"A Declarative System for Optimizing AI Workloads","date":"2024-05-23","arxiv_id":"2405.14696","n_code_links":1,"syntology":null},{"paper":"/paper/agile-a-novel-framework-of-llm-agents","slug":"agile-a-novel-framework-of-llm-agents","title":"AGILE: A Novel Reinforcement Learning Framework of LLM Agents","date":"2024-05-23","arxiv_id":"2405.14751","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bytarnish/agile"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/autocoder-enhancing-code-large-language-model","slug":"autocoder-enhancing-code-large-language-model","title":"AutoCoder: Enhancing Code Large Language Model with \\textsc{AIEV-Instruct}","date":"2024-05-23","arxiv_id":"2405.14906","n_code_links":1,"syntology":null},{"paper":"/paper/deepseek-prover-advancing-theorem-proving-in","slug":"deepseek-prover-advancing-theorem-proving-in","title":"DeepSeek-Prover: Advancing Theorem Proving in LLMs through Large-Scale Synthetic Data","date":"2024-05-23","arxiv_id":"2405.14333","n_code_links":0,"syntology":null},{"paper":"/paper/efficient-medical-question-answering-with","slug":"efficient-medical-question-answering-with","title":"Efficient Medical Question Answering with Knowledge-Augmented Question Generation","date":"2024-05-23","arxiv_id":"2405.14654","n_code_links":1,"syntology":null},{"paper":null,"slug":"evaluating-large-language-models-for-public","title":"Evaluating Large Language Models for Public Health Classification and Extraction Tasks","date":"2024-05-23","arxiv_id":"2405.14766","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-use-of-a-large-language-model","title":"Exploring the use of a Large Language Model for data extraction in systematic reviews: a rapid feasibility study","date":"2024-05-23","arxiv_id":"2405.14445","n_code_links":0,"syntology":null},{"paper":"/paper/impact-of-non-standard-unicode-characters-on","slug":"impact-of-non-standard-unicode-characters-on","title":"Impact of Non-Standard Unicode Characters on Security and Comprehension in Large Language Models","date":"2024-05-23","arxiv_id":"2405.14490","n_code_links":1,"syntology":null},{"paper":null,"slug":"improving-language-models-trained-with","title":"Improving Language Models Trained on Translated Data with Continual Pre-Training and Dictionary Learning Analysis","date":"2024-05-23","arxiv_id":"2405.14277","n_code_links":0,"syntology":null},{"paper":"/paper/jiuzhang3-0-efficiently-improving","slug":"jiuzhang3-0-efficiently-improving","title":"JiuZhang3.0: Efficiently Improving Mathematical Reasoning by Training Small Data Synthesis Models","date":"2024-05-23","arxiv_id":"2405.14365","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["rucaibox/jiuzhang3.0"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"perception-of-knowledge-boundary-for-large","title":"Perception of Knowledge Boundary for Large Language Models through Semi-open-ended Question Answering","date":"2024-05-23","arxiv_id":"2405.14383","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-large-language-models-with-human","slug":"evaluating-large-language-models-with-human","title":"Evaluating Large Language Models with Human Feedback: Establishing a Swedish Benchmark","date":"2024-05-22","arxiv_id":"2405.14006","n_code_links":1,"syntology":null},{"paper":"/paper/why-not-transform-chat-large-language-models","slug":"why-not-transform-chat-large-language-models","title":"Why Not Transform Chat Large Language Models to Non-English?","date":"2024-05-22","arxiv_id":"2405.13923","n_code_links":1,"syntology":null},{"paper":null,"slug":"wordgame-efficient-effective-llm-jailbreak","title":"WordGame: Efficient & Effective LLM Jailbreak via Simultaneous Obfuscation in Query and Response","date":"2024-05-22","arxiv_id":"2405.14023","n_code_links":0,"syntology":null},{"paper":null,"slug":"biomedparse-a-biomedical-foundation-model-for","title":"BiomedParse: a biomedical foundation model for image parsing of everything everywhere all at once","date":"2024-05-21","arxiv_id":"2405.12971","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-and-large-language-models-for","title":"Generative AI in Cybersecurity: A Comprehensive Review of LLM Applications and Vulnerabilities","date":"2024-05-21","arxiv_id":"2405.12750","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpt-4-jailbreaks-itself-with-near-perfect","title":"GPT-4 Jailbreaks Itself with Near-Perfect Success Using Self-Explanation","date":"2024-05-21","arxiv_id":"2405.13077","n_code_links":0,"syntology":null},{"paper":null,"slug":"pathocl-path-based-prompt-augmentation-for","title":"PathOCL: Path-Based Prompt Augmentation for OCL Generation with GPT-4","date":"2024-05-21","arxiv_id":"2405.12450","n_code_links":0,"syntology":null},{"paper":"/paper/can-ai-relate-testing-large-language-model","slug":"can-ai-relate-testing-large-language-model","title":"Can AI Relate: Testing Large Language Model Response for Mental Health Support","date":"2024-05-20","arxiv_id":"2405.12021","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["skgabriel/mh-eval"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ct-eval-benchmarking-chinese-text-to-table","title":"CT-Eval: Benchmarking Chinese Text-to-Table Performance in Large Language Models","date":"2024-05-20","arxiv_id":"2405.12174","n_code_links":0,"syntology":null},{"paper":"/paper/fennec-fine-grained-language-model-evaluation","slug":"fennec-fine-grained-language-model-evaluation","title":"Fennec: Fine-grained Language Model Evaluation and Correction Extended through Branching and Bridging","date":"2024-05-20","arxiv_id":"2405.12163","n_code_links":1,"syntology":null},{"paper":null,"slug":"metacognitive-capabilities-of-llms-an","title":"Metacognitive Capabilities of LLMs: An Exploration in Mathematical Problem Solving","date":"2024-05-20","arxiv_id":"2405.12205","n_code_links":0,"syntology":null},{"paper":null,"slug":"hummer-towards-limited-competitive-preference","title":"Hummer: Towards Limited Competitive Preference Dataset","date":"2024-05-19","arxiv_id":"2405.11647","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-can-infer-personality","title":"Large Language Models Can Infer Personality from Free-Form User Interactions","date":"2024-05-19","arxiv_id":"2405.13052","n_code_links":0,"syntology":null},{"paper":"/paper/mhpp-exploring-the-capabilities-and","slug":"mhpp-exploring-the-capabilities-and","title":"MHPP: Exploring the Capabilities and Limitations of Language Models Beyond Basic Code Generation","date":"2024-05-19","arxiv_id":"2405.11430","n_code_links":1,"syntology":null},{"paper":null,"slug":"automating-ptsd-diagnostics-in-clinical","title":"Automating PTSD Diagnostics in Clinical Interviews: Leveraging Large Language Models for Trauma Assessments","date":"2024-05-18","arxiv_id":"2405.11178","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-public-llms-be-used-for-self-diagnosis-of","title":"Can Public LLMs be used for Self-Diagnosis of Medical Conditions ?","date":"2024-05-18","arxiv_id":"2405.11407","n_code_links":0,"syntology":null},{"paper":null,"slug":"activellm-large-language-model-based-active","title":"ActiveLLM: Large Language Model-based Active Learning for Textual Few-Shot Scenarios","date":"2024-05-17","arxiv_id":"2405.10808","n_code_links":0,"syntology":null},{"paper":null,"slug":"are-large-language-models-moral-hypocrites-a","title":"Are Large Language Models Moral Hypocrites? A Study Based on Moral Foundations","date":"2024-05-17","arxiv_id":"2405.11100","n_code_links":0,"syntology":null},{"paper":"/paper/benchmarking-large-language-models-on-cflue-a","slug":"benchmarking-large-language-models-on-cflue-a","title":"Benchmarking Large Language Models on CFLUE -- A Chinese Financial Language Understanding Evaluation Dataset","date":"2024-05-17","arxiv_id":"2405.10542","n_code_links":2,"syntology":null},{"paper":null,"slug":"enhancing-dialogue-state-tracking-models","title":"Enhancing Dialogue State Tracking Models through LLM-backed User-Agents Simulation","date":"2024-05-17","arxiv_id":"2405.13037","n_code_links":0,"syntology":null},{"paper":"/paper/evaluation-of-large-language-model","slug":"evaluation-of-large-language-model","title":"Evaluation of large language model performance on the Biomedical Language Understanding and Reasoning Benchmark","date":"2024-05-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/language-models-can-evaluate-themselves-via","slug":"language-models-can-evaluate-themselves-via","title":"Language Models can Evaluate Themselves via Probability Discrepancy","date":"2024-05-17","arxiv_id":"2405.10516","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-models-in-wireless-application","title":"Large Language Models in Wireless Application Design: In-Context Learning-enhanced Automatic Network Intrusion Detection","date":"2024-05-17","arxiv_id":"2405.11002","n_code_links":0,"syntology":null},{"paper":"/paper/observational-scaling-laws-and-the","slug":"observational-scaling-laws-and-the","title":"Observational Scaling Laws and the Predictability of Language Model Performance","date":"2024-05-17","arxiv_id":"2405.10938","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ryoungj/obsscaling"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/dynamic-in-context-learning-with","slug":"dynamic-in-context-learning-with","title":"Dynamic In-context Learning with Conversational Models for Data Extraction and Materials Property Prediction","date":"2024-05-16","arxiv_id":"2405.10448","n_code_links":1,"syntology":null},{"paper":null,"slug":"fintextqa-a-dataset-for-long-form-financial","title":"FinTextQA: A Dataset for Long-form Financial Question Answering","date":"2024-05-16","arxiv_id":"2405.09980","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-ai-collaborator-bridging-human-ai","title":"The AI Collaborator: Bridging Human-AI Interaction in Educational and Professional Settings","date":"2024-05-16","arxiv_id":"2405.10460","n_code_links":0,"syntology":null},{"paper":null,"slug":"transcript-of-gpt-4-playing-a-rogue-agi-in-a","title":"Transcript of GPT-4 playing a rogue AGI in a Matrix Game","date":"2024-05-16","arxiv_id":"2405.10997","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-the-efficacy-of-gpt-4-and-chat-gpt","title":"Comparing the Efficacy of GPT-4 and Chat-GPT in Mental Health Care: A Blind Assessment of Large Language Models for Psychological Support","date":"2024-05-15","arxiv_id":"2405.09300","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-potential-of-large-language-8","title":"Exploring the Potential of Large Language Models for Automation in Technical Customer Service","date":"2024-05-15","arxiv_id":"2405.09161","n_code_links":0,"syntology":null},{"paper":null,"slug":"intelligent-tutor-leveraging-chatgpt-and","title":"Intelligent Tutor: Leveraging ChatGPT and Microsoft Copilot Studio to Deliver a Generative AI Student Support and Feedback System within Teams","date":"2024-05-15","arxiv_id":"2405.13024","n_code_links":0,"syntology":null},{"paper":null,"slug":"simulating-policy-impacts-developing-a","title":"Simulating Policy Impacts: Developing a Generative Scenario Writing Method to Evaluate the Perceived Effects of Regulation","date":"2024-05-15","arxiv_id":"2405.09679","n_code_links":0,"syntology":null},{"paper":null,"slug":"sql-to-schema-enhances-schema-linking-in-text","title":"SQL-to-Schema Enhances Schema Linking in Text-to-SQL","date":"2024-05-15","arxiv_id":"2405.09593","n_code_links":0,"syntology":null},{"paper":"/paper/tell-me-why-explainable-public-health-fact","slug":"tell-me-why-explainable-public-health-fact","title":"Tell Me Why: Explainable Public Health Fact-Checking with Large Language Models","date":"2024-05-15","arxiv_id":"2405.09454","n_code_links":1,"syntology":null},{"paper":null,"slug":"word-alignment-as-preference-for-machine","title":"Word Alignment as Preference for Machine Translation","date":"2024-05-15","arxiv_id":"2405.09223","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comprehensive-survey-of-large-language","title":"A Comprehensive Survey of Large Language Models and Multimodal Large Language Models in Medicine","date":"2024-05-14","arxiv_id":"2405.08603","n_code_links":0,"syntology":null},{"paper":"/paper/can-language-models-explain-their-own","slug":"can-language-models-explain-their-own","title":"Can Language Models Explain Their Own Classification Behavior?","date":"2024-05-13","arxiv_id":"2405.07436","n_code_links":1,"syntology":null},{"paper":"/paper/coding-historical-causes-of-death-data-with","slug":"coding-historical-causes-of-death-data-with","title":"Coding historical causes of death data with Large Language Models","date":"2024-05-13","arxiv_id":"2405.07560","n_code_links":1,"syntology":null},{"paper":null,"slug":"metareflection-learning-instructions-for","title":"MetaReflection: Learning Instructions for Language Agents using Past Reflections","date":"2024-05-13","arxiv_id":"2405.13009","n_code_links":0,"syntology":null},{"paper":null,"slug":"plot2code-a-comprehensive-benchmark-for","title":"Plot2Code: A Comprehensive Benchmark for Evaluating Multi-modal Large Language Models in Code Generation from Scientific Plots","date":"2024-05-13","arxiv_id":"2405.07990","n_code_links":0,"syntology":null},{"paper":"/paper/quantifying-and-optimizing-global","slug":"quantifying-and-optimizing-global","title":"Quantifying and Optimizing Global Faithfulness in Persona-driven Role-playing","date":"2024-05-13","arxiv_id":"2405.07726","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["KomeijiForce/Active_Passive_Constraint_Koishiday_2024"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sambanova-sn40l-scaling-the-ai-memory-wall","title":"SambaNova SN40L: Scaling the AI Memory Wall with Dataflow and Composition of Experts","date":"2024-05-13","arxiv_id":"2405.07518","n_code_links":0,"syntology":null}],"record_sha256":"f4a4369e0b2fa16fa05e9c8aa0d6a1716add64bd63427375f62fac0b880bc4c5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}