{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-3/papers/8","list_of":"/method/gpt-3","method":"GPT-3","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":8,"pages_in_order":20,"rows_per_page":100,"rows":[701,800],"of":1906,"counts":{"archive_papers_tagged":1906,"with_a_code_link":866,"where_syntology_ran_a_sample":319,"not_listed_spam_title":0,"listed":1906,"listed_where_code_ran":319,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":259,"every_run_a_failure_of_syntologys_instrument":60,"listed_with_a_run_with_no_instrument_failure":259,"listed_every_run_a_failure_of_syntologys_instrument":60,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-3","prev":"/method/gpt-3/papers/7","next":"/method/gpt-3/papers/9","papers":[{"paper":"/paper/a-critical-evaluation-of-ai-feedback-for","slug":"a-critical-evaluation-of-ai-feedback-for","title":"A Critical Evaluation of AI Feedback for Aligning Large Language Models","date":"2024-02-19","arxiv_id":"2402.12366","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":8,"n_instrument":1,"unverified":2,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["architsharma97/dpo-rlaif"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ask-optimal-questions-aligning-large-language","title":"Ask Optimal Questions: Aligning Large Language Models with Retriever's Preference in Conversational Search","date":"2024-02-19","arxiv_id":"2402.11827","n_code_links":0,"syntology":null},{"paper":null,"slug":"deepcode-ai-fix-fixing-security","title":"DeepCode AI Fix: Fixing Security Vulnerabilities with Large Language Models","date":"2024-02-19","arxiv_id":"2402.13291","n_code_links":0,"syntology":null},{"paper":null,"slug":"end-to-end-multilingual-fact-checking-at","title":"Surprising Efficacy of Fine-Tuned Transformers for Fact-Checking over Larger Language Models","date":"2024-02-19","arxiv_id":"2402.12147","n_code_links":0,"syntology":null},{"paper":null,"slug":"is-open-source-there-yet-a-comparative-study","title":"Is Open-Source There Yet? A Comparative Study on Commercial and Open-Source LLMs in Their Ability to Label Chest X-Ray Reports","date":"2024-02-19","arxiv_id":"2402.12298","n_code_links":0,"syntology":null},{"paper":null,"slug":"meta-ranking-less-capable-language-models-are","title":"Enabling Weak LLMs to Judge Response Reliability via Meta Ranking","date":"2024-02-19","arxiv_id":"2402.12146","n_code_links":0,"syntology":null},{"paper":"/paper/query-based-adversarial-prompt-generation","slug":"query-based-adversarial-prompt-generation","title":"Query-Based Adversarial Prompt Generation","date":"2024-02-19","arxiv_id":"2402.12329","n_code_links":2,"syntology":null},{"paper":null,"slug":"spml-a-dsl-for-defending-language-models","title":"SPML: A DSL for Defending Language Models Against Prompt Attacks","date":"2024-02-19","arxiv_id":"2402.11755","n_code_links":0,"syntology":null},{"paper":null,"slug":"stick-to-your-role-stability-of-personal","title":"Stick to your Role! Stability of Personal Values Expressed in Large Language Models","date":"2024-02-19","arxiv_id":"2402.14846","n_code_links":0,"syntology":null},{"paper":null,"slug":"your-large-language-model-is-secretly-a","title":"Your Large Language Model is Secretly a Fairness Proponent and You Should Prompt it Like One","date":"2024-02-19","arxiv_id":"2402.12150","n_code_links":0,"syntology":null},{"paper":null,"slug":"decoding-news-narratives-a-critical-analysis","title":"Decoding News Narratives: A Critical Analysis of Large Language Models in Framing Detection","date":"2024-02-18","arxiv_id":"2402.11621","n_code_links":0,"syntology":null},{"paper":"/paper/perils-of-self-feedback-self-bias-amplifies","slug":"perils-of-self-feedback-self-bias-amplifies","title":"Pride and Prejudice: LLM Amplifies Self-Bias in Self-Refinement","date":"2024-02-18","arxiv_id":"2402.11436","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xu1998hz/llm_self_bias"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"exploring-chatgpt-for-next-generation","title":"Exploring ChatGPT for Next-generation Information Retrieval: Opportunities and Challenges","date":"2024-02-17","arxiv_id":"2402.11203","n_code_links":0,"syntology":null},{"paper":null,"slug":"gendec-a-robust-generative-question","title":"GenDec: A robust generative Question-decomposition method for Multi-hop reasoning","date":"2024-02-17","arxiv_id":"2402.11166","n_code_links":0,"syntology":null},{"paper":null,"slug":"assessing-the-reasoning-abilities-of-chatgpt","title":"Assessing the Reasoning Abilities of ChatGPT in the Context of Claim Verification","date":"2024-02-16","arxiv_id":"2402.10735","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-separators-improve-chain-of-thought","title":"Can Separators Improve Chain-of-Thought Prompting?","date":"2024-02-16","arxiv_id":"2402.10645","n_code_links":0,"syntology":null},{"paper":"/paper/disordered-dabs-a-benchmark-for-dynamic","slug":"disordered-dabs-a-benchmark-for-dynamic","title":"Disordered-DABS: A Benchmark for Dynamic Aspect-Based Summarization in Disordered Texts","date":"2024-02-16","arxiv_id":"2402.10554","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-as-zero-shot-dialogue","slug":"large-language-models-as-zero-shot-dialogue","title":"Large Language Models as Zero-shot Dialogue State Tracker through Function Calling","date":"2024-02-16","arxiv_id":"2402.10466","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["facebookresearch/fnctod"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"large-language-models-fall-short","title":"Large Language Models Fall Short: Understanding Complex Relationships in Detective Narratives","date":"2024-02-16","arxiv_id":"2402.11051","n_code_links":0,"syntology":null},{"paper":null,"slug":"llms-in-the-heart-of-differential-testing-a","title":"LLMs in the Heart of Differential Testing: A Case Study on a Medical Rule Engine","date":"2024-02-16","arxiv_id":"2404.03664","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-cultural-commonsense-knowledge","title":"Cultural Commonsense Knowledge for Intercultural Dialogues","date":"2024-02-16","arxiv_id":"2402.10689","n_code_links":0,"syntology":null},{"paper":"/paper/universal-prompt-optimizer-for-safe-text-to","slug":"universal-prompt-optimizer-for-safe-text-to","title":"Universal Prompt Optimizer for Safe Text-to-Image Generation","date":"2024-02-16","arxiv_id":"2402.10882","n_code_links":1,"syntology":null},{"paper":null,"slug":"an-analysis-of-langauge-frequency-and-error","title":"An Analysis of Language Frequency and Error Correction for Esperanto","date":"2024-02-15","arxiv_id":"2402.09696","n_code_links":0,"syntology":null},{"paper":null,"slug":"fine-tuning-large-language-model-llm","title":"Fine-tuning Large Language Model (LLM) Artificial Intelligence Chatbots in Ophthalmology and LLM-based evaluation using GPT-4","date":"2024-02-15","arxiv_id":"2402.10083","n_code_links":0,"syntology":null},{"paper":"/paper/pal-proxy-guided-black-box-attack-on-large","slug":"pal-proxy-guided-black-box-attack-on-large","title":"PAL: Proxy-Guided Black-Box Attack on Large Language Models","date":"2024-02-15","arxiv_id":"2402.09674","n_code_links":1,"syntology":null},{"paper":"/paper/the-butterfly-effect-of-model-editing-few","slug":"the-butterfly-effect-of-model-editing-few","title":"The Butterfly Effect of Model Editing: Few Edits Can Trigger Large Language Models Collapse","date":"2024-02-15","arxiv_id":"2402.09656","n_code_links":1,"syntology":null},{"paper":"/paper/api-pack-a-massive-multilingual-dataset-for","slug":"api-pack-a-massive-multilingual-dataset-for","title":"API Pack: A Massive Multi-Programming Language Dataset for API Call Generation","date":"2024-02-14","arxiv_id":"2402.09615","n_code_links":1,"syntology":null},{"paper":"/paper/mpirigen-mpi-code-generation-through-domain","slug":"mpirigen-mpi-code-generation-through-domain","title":"MPIrigen: MPI Code Generation through Domain-Specific Language Models","date":"2024-02-14","arxiv_id":"2402.09126","n_code_links":1,"syntology":null},{"paper":null,"slug":"auditing-counterfire-evaluating-advanced","title":"\"Reasoning\" with Rhetoric: On the Style-Evidence Tradeoff in LLM-Generated Counter-Arguments","date":"2024-02-13","arxiv_id":"2402.08498","n_code_links":0,"syntology":null},{"paper":"/paper/cold-attack-jailbreaking-llms-with","slug":"cold-attack-jailbreaking-llms-with","title":"COLD-Attack: Jailbreaking LLMs with Stealthiness and Controllability","date":"2024-02-13","arxiv_id":"2402.08679","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":7,"n_instrument":0,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yu-fangxu/cold-attack"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"lying-blindly-bypassing-chatgpt-s-safeguards","title":"Lying Blindly: Bypassing ChatGPT's Safeguards to Generate Hard-to-Detect Disinformation Claims","date":"2024-02-13","arxiv_id":"2402.08467","n_code_links":0,"syntology":null},{"paper":"/paper/measuring-and-controlling-instruction-in","slug":"measuring-and-controlling-instruction-in","title":"Measuring and Controlling Instruction (In)Stability in Language Model Dialogs","date":"2024-02-13","arxiv_id":"2402.10962","n_code_links":1,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["likenneth/persona_drift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"mitigating-object-hallucination-in-large","title":"Mitigating Object Hallucination in Large Vision-Language Models via Classifier-Free Guidance","date":"2024-02-13","arxiv_id":"2402.08680","n_code_links":0,"syntology":null},{"paper":"/paper/prompt-optimization-in-multi-step-tasks","slug":"prompt-optimization-in-multi-step-tasks","title":"PRompt Optimization in Multi-Step Tasks (PROMST): Integrating Human Feedback and Heuristic-based Sampling","date":"2024-02-13","arxiv_id":"2402.08702","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yongchao98/promst"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/addressing-cognitive-bias-in-medical-language","slug":"addressing-cognitive-bias-in-medical-language","title":"Addressing cognitive bias in medical language models","date":"2024-02-12","arxiv_id":"2402.08113","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["carlwharris/cog-bias-med-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/breakgpt-a-large-language-model-with-multi","slug":"breakgpt-a-large-language-model-with-multi","title":"BreakGPT: A Large Language Model with Multi-stage Structure for Financial Breakout Detection","date":"2024-02-12","arxiv_id":"2402.07536","n_code_links":1,"syntology":null},{"paper":"/paper/cybermetric-a-benchmark-dataset-for","slug":"cybermetric-a-benchmark-dataset-for","title":"CyberMetric: A Benchmark Dataset based on Retrieval-Augmented Generation for Evaluating LLMs in Cybersecurity Knowledge","date":"2024-02-12","arxiv_id":"2402.07688","n_code_links":1,"syntology":null},{"paper":null,"slug":"investigating-the-impact-of-data","title":"Investigating the Impact of Data Contamination of Large Language Models in Text-to-SQL Translation","date":"2024-02-12","arxiv_id":"2402.08100","n_code_links":0,"syntology":null},{"paper":null,"slug":"sequential-ordering-in-textual-descriptions","title":"Can Graph Descriptive Order Affect Solving Graph Problems with LLMs?","date":"2024-02-11","arxiv_id":"2402.07140","n_code_links":0,"syntology":null},{"paper":"/paper/chemllm-a-chemical-large-language-model","slug":"chemllm-a-chemical-large-language-model","title":"ChemLLM: A Chemical Large Language Model","date":"2024-02-10","arxiv_id":"2402.06852","n_code_links":1,"syntology":null},{"paper":"/paper/culturellm-incorporating-cultural-differences","slug":"culturellm-incorporating-cultural-differences","title":"CultureLLM: Incorporating Cultural Differences into Large Language Models","date":"2024-02-09","arxiv_id":"2402.10946","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["scarelette/culturellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/exaranker-open-synthetic-explanation-for-ir","slug":"exaranker-open-synthetic-explanation-for-ir","title":"ExaRanker-Open: Synthetic Explanation for IR using Open-Source LLMs","date":"2024-02-09","arxiv_id":"2402.06334","n_code_links":1,"syntology":null},{"paper":"/paper/comprehensive-assessment-of-jailbreak-attacks","slug":"comprehensive-assessment-of-jailbreak-attacks","title":"JailbreakRadar: Comprehensive Assessment of Jailbreak Attacks Against LLMs","date":"2024-02-08","arxiv_id":"2402.05668","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["TrustAIRLab/Comprehensive_Jailbreak_Assessment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/in-context-principle-learning-from-mistakes","slug":"in-context-principle-learning-from-mistakes","title":"In-Context Principle Learning from Mistakes","date":"2024-02-08","arxiv_id":"2402.05403","n_code_links":1,"syntology":null},{"paper":null,"slug":"zero-shot-chain-of-thought-reasoning-guided","title":"Zero-Shot Chain-of-Thought Reasoning Guided by Evolutionary Algorithms in Large Language Models","date":"2024-02-08","arxiv_id":"2402.05376","n_code_links":0,"syntology":null},{"paper":"/paper/a-hypothesis-driven-framework-for-the","slug":"a-hypothesis-driven-framework-for-the","title":"A Hypothesis-Driven Framework for the Analysis of Self-Rationalising Models","date":"2024-02-07","arxiv_id":"2402.04787","n_code_links":1,"syntology":null},{"paper":"/paper/grandmaster-level-chess-without-search","slug":"grandmaster-level-chess-without-search","title":"Amortized Planning with Large-Scale Transformers: A Case Study on Chess","date":"2024-02-07","arxiv_id":"2402.04494","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["google-deepmind/searchless_chess"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-cross-domain-low-resource-text","title":"Improving Cross-Domain Low-Resource Text Generation through LLM Post-Editing: A Programmer-Interpreter Approach","date":"2024-02-07","arxiv_id":"2402.04609","n_code_links":0,"syntology":null},{"paper":"/paper/long-is-more-for-alignment-a-simple-but-tough","slug":"long-is-more-for-alignment-a-simple-but-tough","title":"Long Is More for Alignment: A Simple but Tough-to-Beat Baseline for Instruction Fine-Tuning","date":"2024-02-07","arxiv_id":"2402.04833","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/long-is-more-for-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"behind-the-screen-investigating-chatgpt-s","title":"Behind the Screen: Investigating ChatGPT's Dark Personality Traits and Conspiracy Beliefs","date":"2024-02-06","arxiv_id":"2402.04110","n_code_links":0,"syntology":null},{"paper":null,"slug":"detecting-mode-collapse-in-language-models","title":"Detecting Mode Collapse in Language Models via Narration","date":"2024-02-06","arxiv_id":"2402.04477","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-an-indirect-reasoner","title":"Large Language Models as an Indirect Reasoner: Contrapositive and Contradiction for Automated Reasoning","date":"2024-02-06","arxiv_id":"2402.03667","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-as-moocs-graders","title":"Large Language Models As MOOCs Graders","date":"2024-02-06","arxiv_id":"2402.03776","n_code_links":0,"syntology":null},{"paper":null,"slug":"leak-cheat-repeat-data-contamination-and","title":"Leak, Cheat, Repeat: Data Contamination and Evaluation Malpractices in Closed-Source LLMs","date":"2024-02-06","arxiv_id":"2402.03927","n_code_links":0,"syntology":null},{"paper":null,"slug":"minds-versus-machines-rethinking-entailment","title":"Are Machines Better at Complex Reasoning? Unveiling Human-Machine Inference Gaps in Entailment Verification","date":"2024-02-06","arxiv_id":"2402.03686","n_code_links":0,"syntology":null},{"paper":"/paper/training-language-models-to-generate-text","slug":"training-language-models-to-generate-text","title":"Training Language Models to Generate Text with Citations via Fine-grained Rewards","date":"2024-02-06","arxiv_id":"2402.04315","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":3,"n_instrument":4,"unverified":5,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","official":{"repos":["hcy123902/atg-w-fg-rw"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/conversation-reconstruction-attack-against","slug":"conversation-reconstruction-attack-against","title":"Reconstruct Your Previous Conversations! Comprehensively Investigating Privacy Leakage Risks in Conversations with GPT Models","date":"2024-02-05","arxiv_id":"2402.02987","n_code_links":1,"syntology":null},{"paper":null,"slug":"harnessing-pubmed-user-query-logs-for-post","title":"Harnessing PubMed User Query Logs for Post Hoc Explanations of Recommended Similar Articles","date":"2024-02-05","arxiv_id":"2402.03484","n_code_links":0,"syntology":null},{"paper":"/paper/llm-agents-in-interaction-measuring","slug":"llm-agents-in-interaction-measuring","title":"LLM Agents in Interaction: Measuring Personality Consistency and Linguistic Alignment in Interacting Populations of Large Language Models","date":"2024-02-05","arxiv_id":"2402.02896","n_code_links":1,"syntology":null},{"paper":"/paper/swag-storytelling-with-action-guidance","slug":"swag-storytelling-with-action-guidance","title":"SWAG: Storytelling With Action Guidance","date":"2024-02-05","arxiv_id":"2402.03483","n_code_links":1,"syntology":null},{"paper":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","n_code_links":1,"syntology":{"ran":13,"of":18,"n_ran_checked":13,"n_instrument":0,"unverified":5,"pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-assessment-of-tutoring-practices","title":"Improving Assessment of Tutoring Practices using Retrieval-Augmented Generation","date":"2024-02-04","arxiv_id":"2402.14594","n_code_links":0,"syntology":null},{"paper":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":1,"n_instrument":2,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-the-limitations-of-graph-reasoning","slug":"exploring-the-limitations-of-graph-reasoning","title":"Can LLMs perform structured graph reasoning?","date":"2024-02-02","arxiv_id":"2402.01805","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-planning-based-reasoning-by","title":"Learning Planning-based Reasoning by Trajectories Collection and Process Reward Synthesizing","date":"2024-02-01","arxiv_id":"2402.00658","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-contrastive-pre-training-for-1","title":"Self-Supervised Contrastive Pre-Training for Multivariate Point Processes","date":"2024-02-01","arxiv_id":"2402.00987","n_code_links":0,"syntology":null},{"paper":null,"slug":"tiny-titans-can-smaller-large-language-models","title":"Tiny Titans: Can Smaller Large Language Models Punch Above Their Weight in the Real World for Meeting Summarization?","date":"2024-02-01","arxiv_id":"2402.00841","n_code_links":0,"syntology":null},{"paper":null,"slug":"global-liar-factuality-of-llms-over-time-and","title":"Global-Liar: Factuality of LLMs over Time and Geographic Regions","date":"2024-01-31","arxiv_id":"2401.17839","n_code_links":0,"syntology":null},{"paper":null,"slug":"making-a-long-story-short-in-conversation","title":"Making a Long Story Short in Conversation Modeling","date":"2024-01-31","arxiv_id":"2402.00143","n_code_links":0,"syntology":null},{"paper":null,"slug":"mitigating-the-problem-of-strong-priors-in","title":"Mitigating the Influence of Distractor Tasks in LMs with Prior-Aware Decoding","date":"2024-01-31","arxiv_id":"2401.17692","n_code_links":0,"syntology":null},{"paper":null,"slug":"paramanu-a-family-of-novel-efficient-indic","title":"Paramanu: A Family of Novel Efficient Generative Foundation Language Models for Indian Languages","date":"2024-01-31","arxiv_id":"2401.18034","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-preliminary-study-on-using-large-language","title":"A Preliminary Study on Using Large Language Models in Software Pentesting","date":"2024-01-30","arxiv_id":"2401.17459","n_code_links":0,"syntology":null},{"paper":"/paper/llamp-large-language-model-made-powerful-for","slug":"llamp-large-language-model-made-powerful-for","title":"LLaMP: Large Language Model Made Powerful for High-fidelity Materials Knowledge Retrieval and Distillation","date":"2024-01-30","arxiv_id":"2401.17244","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chiang-yuan/llamp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mt-eval-a-multi-turn-capabilities-evaluation","slug":"mt-eval-a-multi-turn-capabilities-evaluation","title":"MT-Eval: A Multi-Turn Capabilities Evaluation Benchmark for Large Language Models","date":"2024-01-30","arxiv_id":"2401.16745","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["kwanwaichung/mt-eval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"leveraging-professional-radiologists","title":"Leveraging Professional Radiologists' Expertise to Enhance LLMs' Evaluation for Radiology Reports","date":"2024-01-29","arxiv_id":"2401.16578","n_code_links":0,"syntology":null},{"paper":null,"slug":"llm4vuln-a-unified-evaluation-framework-for","title":"LLM4Vuln: A Unified Evaluation Framework for Decoupling and Enhancing LLMs' Vulnerability Reasoning","date":"2024-01-29","arxiv_id":"2401.16185","n_code_links":0,"syntology":null},{"paper":"/paper/regal-refactoring-programs-to-discover","slug":"regal-refactoring-programs-to-discover","title":"ReGAL: Refactoring Programs to Discover Generalizable Abstractions","date":"2024-01-29","arxiv_id":"2401.16467","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["esteng/regal_program_learning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"security-code-review-by-llms-a-deep-dive-into","title":"An Insight into Security Code Review with LLMs: Capabilities, Obstacles, and Influential Factors","date":"2024-01-29","arxiv_id":"2401.16310","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-large-language-model-performance-to","title":"Enhancing Large Language Model Performance To Answer Questions and Extract Information More Accurately","date":"2024-01-27","arxiv_id":"2402.01722","n_code_links":0,"syntology":null},{"paper":null,"slug":"equipping-language-models-with-tool-use","title":"Equipping Language Models with Tool Use Capability for Tabular Data Analysis in Finance","date":"2024-01-27","arxiv_id":"2401.15328","n_code_links":0,"syntology":null},{"paper":null,"slug":"fortifying-ethical-boundaries-in-ai-advanced","title":"Fortifying Ethical Boundaries in AI: Advanced Strategies for Enhancing Security in Large Language Models","date":"2024-01-27","arxiv_id":"2402.01725","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-qualitative-coding-with-llms-chain","title":"Scalable Qualitative Coding with LLMs: Chain-of-Thought Reasoning Matches Human Performance in Some Hermeneutic Tasks","date":"2024-01-26","arxiv_id":"2401.15170","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-comparative-study-of-zero-shot-inference","title":"A comparative study of zero-shot inference with large language models and supervised modeling in breast cancer pathology classification","date":"2024-01-25","arxiv_id":"2401.13887","n_code_links":0,"syntology":null},{"paper":"/paper/deepseek-coder-when-the-large-language-model","slug":"deepseek-coder-when-the-large-language-model","title":"DeepSeek-Coder: When the Large Language Model Meets Programming -- The Rise of Code Intelligence","date":"2024-01-25","arxiv_id":"2401.14196","n_code_links":1,"syntology":{"ran":9,"of":10,"n_ran_checked":9,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["deepseek-ai/DeepSeek-Coder"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluating-gpt-3-5-s-awareness-and","title":"Evaluating GPT-3.5's Awareness and Summarization Abilities for European Constitutional Texts with Shared Topics","date":"2024-01-25","arxiv_id":"2401.14524","n_code_links":0,"syntology":null},{"paper":null,"slug":"investigate-consolidate-exploit-a-general","title":"Investigate-Consolidate-Exploit: A General Strategy for Inter-Task Agent Self-Evolution","date":"2024-01-25","arxiv_id":"2401.13996","n_code_links":0,"syntology":null},{"paper":"/paper/longhealth-a-question-answering-benchmark","slug":"longhealth-a-question-answering-benchmark","title":"LongHealth: A Question Answering Benchmark with Long Clinical Documents","date":"2024-01-25","arxiv_id":"2401.14490","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kbressem/longhealth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/tricy-trigger-guided-data-to-text-generation-1","slug":"tricy-trigger-guided-data-to-text-generation-1","title":"TrICy: Trigger-guided Data-to-text Generation with Intent aware Attention-Copy","date":"2024-01-25","arxiv_id":"2402.01714","n_code_links":0,"syntology":null},{"paper":null,"slug":"unmasking-and-quantifying-racial-bias-of","title":"Unmasking and Quantifying Racial Bias of Large Language Models in Medical Report Generation","date":"2024-01-25","arxiv_id":"2401.13867","n_code_links":0,"syntology":null},{"paper":null,"slug":"zs4c-zero-shot-synthesis-of-compilable-code","title":"ZS4C: Zero-Shot Synthesis of Compilable Code for Incomplete Code Snippets using LLMs","date":"2024-01-25","arxiv_id":"2401.14279","n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-root-causing-of-cloud-incidents","title":"Automated Root Causing of Cloud Incidents using In-Context Learning with GPT-4","date":"2024-01-24","arxiv_id":"2401.13810","n_code_links":0,"syntology":null},{"paper":"/paper/can-gpt-3-5-generate-and-code-discharge","slug":"can-gpt-3-5-generate-and-code-discharge","title":"Can GPT-3.5 Generate and Code Discharge Summaries?","date":"2024-01-24","arxiv_id":"2401.13512","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["edinburghclinicalnlp/chatgpt_icd_coding"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"evaluation-of-general-large-language-models","title":"Evaluation of General Large Language Models in Contextually Assessing Semantic Concepts Extracted from Adult Critical Care Electronic Health Record Notes","date":"2024-01-24","arxiv_id":"2401.13588","n_code_links":0,"syntology":null},{"paper":null,"slug":"kam-cot-knowledge-augmented-multimodal-chain","title":"KAM-CoT: Knowledge Augmented Multimodal Chain-of-Thoughts Reasoning","date":"2024-01-23","arxiv_id":"2401.12863","n_code_links":0,"syntology":null},{"paper":"/paper/badchain-backdoor-chain-of-thought-prompting","slug":"badchain-backdoor-chain-of-thought-prompting","title":"BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models","date":"2024-01-20","arxiv_id":"2401.12242","n_code_links":1,"syntology":{"ran":13,"of":15,"n_ran_checked":13,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["django-jiang/badchain"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-large-language-models-for-clinical","title":"Enhancing Large Language Models for Clinical Decision Support by Incorporating Clinical Practice Guidelines","date":"2024-01-20","arxiv_id":"2401.11120","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-and-enhancing-large-language","title":"Evaluating and Enhancing Large Language Models Performance in Domain-specific Medicine: Osteoarthritis Management with DocOA","date":"2024-01-20","arxiv_id":"2401.12998","n_code_links":0,"syntology":null},{"paper":null,"slug":"finllms-a-framework-for-financial-reasoning","title":"FinLLMs: A Framework for Financial Reasoning Dataset Generation with Large Language Models","date":"2024-01-19","arxiv_id":"2401.10744","n_code_links":0,"syntology":null},{"paper":"/paper/mining-experimental-data-from-materials","slug":"mining-experimental-data-from-materials","title":"Mining experimental data from Materials Science literature with Large Language Models: an evaluation study","date":"2024-01-19","arxiv_id":"2401.11052","n_code_links":1,"syntology":null},{"paper":null,"slug":"gender-bias-in-machine-translation-and-the","title":"Gender Bias in Machine Translation and The Era of Large Language Models","date":"2024-01-18","arxiv_id":"2401.10016","n_code_links":0,"syntology":null}],"record_sha256":"fe3658f773e7132f3bae07b2beb659b113b457c1660c9082dd332f5115211dbf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}