{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/6","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":6,"pages_in_order":29,"rows_per_page":100,"rows":[501,600],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/5","next":"/method/gpt-4/papers/7","papers":[{"paper":null,"slug":"llms-a-game-changer-for-software-engineers","title":"LLMs: A Game-Changer for Software Engineers?","date":"2024-11-01","arxiv_id":"2411.00932","n_code_links":0,"syntology":null},{"paper":"/paper/self-evolved-reward-learning-for-llms","slug":"self-evolved-reward-learning-for-llms","title":"Self-Evolved Reward Learning for LLMs","date":"2024-11-01","arxiv_id":"2411.00418","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":null}},{"paper":null,"slug":"desert-camels-and-oil-sheikhs-arab-centric","title":"Desert Camels and Oil Sheikhs: Arab-Centric Red Teaming of Frontier LLMs","date":"2024-10-31","arxiv_id":"2410.24049","n_code_links":0,"syntology":null},{"paper":null,"slug":"large-language-models-for-patient-comments","title":"Large Language Models for Patient Comments Multi-Label Classification","date":"2024-10-31","arxiv_id":"2410.23528","n_code_links":0,"syntology":null},{"paper":"/paper/rsl-sql-robust-schema-linking-in-text-to-sql","slug":"rsl-sql-robust-schema-linking-in-text-to-sql","title":"RSL-SQL: Robust Schema Linking in Text-to-SQL Generation","date":"2024-10-31","arxiv_id":"2411.00073","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["laqcce-cao/rsl-sql"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"danoliteracy-of-generative-large-language","title":"Danoliteracy of Generative, Large Language Models","date":"2024-10-30","arxiv_id":"2410.22839","n_code_links":0,"syntology":null},{"paper":"/paper/evocodebench-an-evolving-code-generation-1","slug":"evocodebench-an-evolving-code-generation-1","title":"EvoCodeBench: An Evolving Code Generation Benchmark with Domain-Specific Evaluations","date":"2024-10-30","arxiv_id":"2410.22821","n_code_links":0,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/scipip-an-llm-based-scientific-paper-idea","slug":"scipip-an-llm-based-scientific-paper-idea","title":"SciPIP: An LLM-based Scientific Paper Idea Proposer","date":"2024-10-30","arxiv_id":"2410.23166","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cheerss/scipip"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/amplegcg-plus-a-strong-generative-model-of","slug":"amplegcg-plus-a-strong-generative-model-of","title":"AmpleGCG-Plus: A Strong Generative Model of Adversarial Suffixes to Jailbreak LLMs with Higher Success Rates in Fewer Attempts","date":"2024-10-29","arxiv_id":"2410.22143","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"cfsafety-comprehensive-fine-grained-safety","title":"CFSafety: Comprehensive Fine-grained Safety Assessment for LLMs","date":"2024-10-29","arxiv_id":"2410.21695","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-preference-bias-in-llm-as-a-judge","title":"Self-Preference Bias in LLM-as-a-Judge","date":"2024-10-29","arxiv_id":"2410.21819","n_code_links":0,"syntology":null},{"paper":"/paper/topic-conversation-relevance-tcr-dataset-and","slug":"topic-conversation-relevance-tcr-dataset-and","title":"Topic-Conversation Relevance (TCR) Dataset and Benchmarks","date":"2024-10-29","arxiv_id":"2411.00038","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["microsoft/topic_conversation"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-simple-yet-effective-corpus-construction-1","slug":"a-simple-yet-effective-corpus-construction-1","title":"A Simple Yet Effective Corpus Construction Framework for Indonesian Grammatical Error Correction","date":"2024-10-28","arxiv_id":"2410.20838","n_code_links":1,"syntology":null},{"paper":"/paper/belief-in-the-machine-investigating","slug":"belief-in-the-machine-investigating","title":"Belief in the Machine: Investigating Epistemological Blind Spots of Language Models","date":"2024-10-28","arxiv_id":"2410.21195","n_code_links":1,"syntology":null},{"paper":null,"slug":"ct2c-qa-multimodal-question-answering-over","title":"CT2C-QA: Multimodal Question Answering over Chinese Text, Table and Chart","date":"2024-10-28","arxiv_id":"2410.21414","n_code_links":0,"syntology":null},{"paper":null,"slug":"gender-bias-in-llm-generated-interview","title":"Gender Bias in LLM-generated Interview Responses","date":"2024-10-28","arxiv_id":"2410.20739","n_code_links":0,"syntology":null},{"paper":"/paper/instruction-tuned-llms-succeed-in-document","slug":"instruction-tuned-llms-succeed-in-document","title":"Fine-Grained and Multi-Dimensional Metrics for Document-Level Machine Translation","date":"2024-10-28","arxiv_id":"2410.20941","n_code_links":1,"syntology":null},{"paper":null,"slug":"is-gpt-4-less-politically-biased-than-gpt-3-5","title":"Is GPT-4 Less Politically Biased than GPT-3.5? A Renewed Investigation of ChatGPT's Political Biases","date":"2024-10-28","arxiv_id":"2410.21008","n_code_links":0,"syntology":null},{"paper":null,"slug":"sandboxaq-s-submission-to-mrl-2024-shared","title":"SandboxAQ's submission to MRL 2024 Shared Task on Multi-lingual Multi-task Information Retrieval","date":"2024-10-28","arxiv_id":"2410.21501","n_code_links":0,"syntology":null},{"paper":null,"slug":"malinowski-in-the-age-of-ai-can-large","title":"Malinowski in the Age of AI: Can large language models create a text game based on an anthropological classic?","date":"2024-10-27","arxiv_id":"2410.20536","n_code_links":0,"syntology":null},{"paper":"/paper/sequential-large-language-model-based-hyper","slug":"sequential-large-language-model-based-hyper","title":"Sequential Large Language Model-Based Hyper-parameter Optimization","date":"2024-10-27","arxiv_id":"2410.20302","n_code_links":1,"syntology":null},{"paper":null,"slug":"think-carefully-and-check-again-meta","title":"Think Carefully and Check Again! Meta-Generation Unlocking LLMs for Low-Resource Cross-Lingual Summarization","date":"2024-10-26","arxiv_id":"2410.20021","n_code_links":0,"syntology":null},{"paper":null,"slug":"fairmt-bench-benchmarking-fairness-for-multi","title":"FairMT-Bench: Benchmarking Fairness for Multi-turn Dialogue in Conversational LLMs","date":"2024-10-25","arxiv_id":"2410.19317","n_code_links":0,"syntology":null},{"paper":"/paper/gpt-4o-system-card","slug":"gpt-4o-system-card","title":"GPT-4o System Card","date":"2024-10-25","arxiv_id":"2410.21276","n_code_links":0,"syntology":null},{"paper":null,"slug":"kahani-culturally-nuanced-visual-storytelling","title":"KAHANI: Culturally-Nuanced Visual Storytelling Pipeline for Non-Western Cultures","date":"2024-10-25","arxiv_id":"2410.19419","n_code_links":0,"syntology":null},{"paper":null,"slug":"robot-behavior-personalization-from-sparse","title":"Robot Behavior Personalization from Sparse User Feedback","date":"2024-10-25","arxiv_id":"2410.19219","n_code_links":0,"syntology":null},{"paper":null,"slug":"aggregated-knowledge-model-enhancing-domain","title":"Aggregated Knowledge Model: Enhancing Domain-Specific QA with Fine-Tuned and Retrieval-Augmented Generation Models","date":"2024-10-24","arxiv_id":"2410.18344","n_code_links":0,"syntology":null},{"paper":null,"slug":"biomistral-nlu-towards-more-generalizable","title":"BioMistral-NLU: Towards More Generalizable Medical Language Understanding through Instruction Tuning","date":"2024-10-24","arxiv_id":"2410.18955","n_code_links":0,"syntology":null},{"paper":"/paper/camel-bench-a-comprehensive-arabic-lmm","slug":"camel-bench-a-comprehensive-arabic-lmm","title":"CAMEL-Bench: A Comprehensive Arabic LMM Benchmark","date":"2024-10-24","arxiv_id":"2410.18976","n_code_links":1,"syntology":null},{"paper":"/paper/iterative-self-tuning-llms-for-enhanced","slug":"iterative-self-tuning-llms-for-enhanced","title":"Iterative Self-Tuning LLMs for Enhanced Jailbreaking Capabilities","date":"2024-10-24","arxiv_id":"2410.18469","n_code_links":1,"syntology":null},{"paper":"/paper/little-giants-synthesizing-high-quality","slug":"little-giants-synthesizing-high-quality","title":"Little Giants: Synthesizing High-Quality Embedding Data at Scale","date":"2024-10-24","arxiv_id":"2410.18634","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haon-chen/SPEED"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/logo-long-context-alignment-via-efficient","slug":"logo-long-context-alignment-via-efficient","title":"LOGO -- Long cOntext aliGnment via efficient preference Optimization","date":"2024-10-24","arxiv_id":"2410.18533","n_code_links":1,"syntology":null},{"paper":null,"slug":"prompting-and-fine-tuning-of-small-llms-for","title":"Prompting and Fine-Tuning of Small LLMs for Length-Controllable Telephone Call Summarization","date":"2024-10-24","arxiv_id":"2410.18624","n_code_links":0,"syntology":null},{"paper":null,"slug":"toolflow-boosting-llm-tool-calling-through","title":"ToolFlow: Boosting LLM Tool-Calling Through Natural and Coherent Dialogue Synthesis","date":"2024-10-24","arxiv_id":"2410.18447","n_code_links":0,"syntology":null},{"paper":null,"slug":"clr-bench-evaluating-large-language-models-in","title":"CLR-Bench: Evaluating Large Language Models in College-level Reasoning","date":"2024-10-23","arxiv_id":"2410.17558","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-pdfs-to-structured-data-utilizing-llm","title":"From PDFs to Structured Data: Utilizing LLM Analysis in Sports Database Management","date":"2024-10-23","arxiv_id":"2410.17619","n_code_links":0,"syntology":null},{"paper":null,"slug":"gazelle-an-instruction-dataset-for-arabic","title":"Gazelle: An Instruction Dataset for Arabic Writing Assistance","date":"2024-10-23","arxiv_id":"2410.18163","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-statistical-analysis-of-llms-self","title":"A Statistical Analysis of LLMs' Self-Evaluation Using Proverbs","date":"2024-10-22","arxiv_id":"2410.16640","n_code_links":0,"syntology":null},{"paper":null,"slug":"an-eye-for-an-ai-evaluating-gpt-4o-s-visual","title":"An Eye for an AI: Evaluating GPT-4o's Visual Perception Skills and Geometric Reasoning Skills Using Computer Graphics Questions","date":"2024-10-22","arxiv_id":"2410.16991","n_code_links":0,"syntology":null},{"paper":"/paper/automated-spinal-mri-labelling-from-reports","slug":"automated-spinal-mri-labelling-from-reports","title":"Automated Spinal MRI Labelling from Reports Using a Large Language Model","date":"2024-10-22","arxiv_id":"2410.17235","n_code_links":1,"syntology":null},{"paper":"/paper/in-context-learning-and-reasoning-for","slug":"in-context-learning-and-reasoning-for","title":"In Context Learning and Reasoning for Symbolic Regression with Large Language Models","date":"2024-10-22","arxiv_id":"2410.17448","n_code_links":1,"syntology":null},{"paper":"/paper/an-efficient-system-for-automatic-map","slug":"an-efficient-system-for-automatic-map","title":"An Efficient System for Automatic Map Storytelling -- A Case Study on Historical Maps","date":"2024-10-21","arxiv_id":"2410.15780","n_code_links":1,"syntology":null},{"paper":"/paper/causalgraph2llm-evaluating-llms-for-causal","slug":"causalgraph2llm-evaluating-llms-for-causal","title":"CausalGraph2LLM: Evaluating LLMs for Causal Queries","date":"2024-10-21","arxiv_id":"2410.15939","n_code_links":1,"syntology":null},{"paper":"/paper/reflection-bench-probing-ai-intelligence-with","slug":"reflection-bench-probing-ai-intelligence-with","title":"Reflection-Bench: probing AI intelligence with reflection","date":"2024-10-21","arxiv_id":"2410.16270","n_code_links":1,"syntology":null},{"paper":null,"slug":"students-rather-than-experts-a-new-ai-for","title":"Students Rather Than Experts: A New AI For Education Pipeline To Model More Human-Like And Personalised Early Adolescences","date":"2024-10-21","arxiv_id":"2410.15701","n_code_links":0,"syntology":null},{"paper":"/paper/back-to-school-translation-using-grammar","slug":"back-to-school-translation-using-grammar","title":"Back to School: Translation Using Grammar Books","date":"2024-10-20","arxiv_id":"2410.15263","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["jonathanhus/back-to-school"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/does-chatgpt-have-a-poetic-style","slug":"does-chatgpt-have-a-poetic-style","title":"Does ChatGPT Have a Poetic Style?","date":"2024-10-20","arxiv_id":"2410.15299","n_code_links":1,"syntology":null},{"paper":null,"slug":"exploring-social-desirability-response-bias","title":"Exploring Social Desirability Response Bias in Large Language Models: Evidence from GPT-4 Simulations","date":"2024-10-20","arxiv_id":"2410.15442","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-language-models-to-critique-with","title":"Training Language Models to Critique With Multi-agent Feedback","date":"2024-10-20","arxiv_id":"2410.15287","n_code_links":0,"syntology":null},{"paper":"/paper/semihvision-enhancing-medical-multimodal","slug":"semihvision-enhancing-medical-multimodal","title":"SemiHVision: Enhancing Medical Multimodal Models with a Semi-Human Annotated Dataset and Fine-Tuned Instruction Generation","date":"2024-10-19","arxiv_id":"2410.14948","n_code_links":1,"syntology":null},{"paper":null,"slug":"causalchat-interactive-causal-model","title":"CausalChat: Interactive Causal Model Development and Refinement Using Large Language Models","date":"2024-10-18","arxiv_id":"2410.14146","n_code_links":0,"syntology":null},{"paper":null,"slug":"celi-controller-embedded-language-model","title":"CELI: Controller-Embedded Language Model Interactions","date":"2024-10-18","arxiv_id":"2410.14627","n_code_links":0,"syntology":null},{"paper":null,"slug":"dflow-diverse-dialogue-flow-simulation-with","title":"DFlow: Diverse Dialogue Flow Simulation with Large Language Models","date":"2024-10-18","arxiv_id":"2410.14853","n_code_links":0,"syntology":null},{"paper":null,"slug":"flame-quality-monitoring-of-flare-stack-based","title":"Flame quality monitoring of flare stack based on deep visual features","date":"2024-10-18","arxiv_id":"2410.19823","n_code_links":0,"syntology":null},{"paper":null,"slug":"good-parenting-is-all-you-need-multi-agentic","title":"Good Parenting is all you need -- Multi-agentic LLM Hallucination Mitigation","date":"2024-10-18","arxiv_id":"2410.14262","n_code_links":0,"syntology":null},{"paper":null,"slug":"novel-development-of-llm-driven-mcode-data","title":"Novel Development of LLM Driven mCODE Data Model for Improved Clinical Trial Matching to Enable Standardization and Interoperability in Oncology Research","date":"2024-10-18","arxiv_id":"2410.19826","n_code_links":0,"syntology":null},{"paper":"/paper/paths-over-graph-knowledge-graph-enpowered","slug":"paths-over-graph-knowledge-graph-enpowered","title":"Paths-over-Graph: Knowledge Graph Empowered Large Language Model Reasoning","date":"2024-10-18","arxiv_id":"2410.14211","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/timeseriesexam-a-time-series-understanding","slug":"timeseriesexam-a-time-series-understanding","title":"TimeSeriesExam: A time series understanding exam","date":"2024-10-18","arxiv_id":"2410.14752","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"better-to-ask-in-english-evaluation-of-large","title":"Better to Ask in English: Evaluation of Large Language Models on English, Low-resource and Cross-Lingual Settings","date":"2024-10-17","arxiv_id":"2410.13153","n_code_links":0,"syntology":null},{"paper":null,"slug":"iterselecttune-an-iterative-training","title":"IterSelectTune: An Iterative Training Framework for Efficient Instruction-Tuning Data Selection","date":"2024-10-17","arxiv_id":"2410.13464","n_code_links":0,"syntology":null},{"paper":"/paper/looking-inward-language-models-can-learn","slug":"looking-inward-language-models-can-learn","title":"Looking Inward: Language Models Can Learn About Themselves by Introspection","date":"2024-10-17","arxiv_id":"2410.13787","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["felixbinder/introspection_self_prediction"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mcqg-srefine-multiple-choice-question","slug":"mcqg-srefine-multiple-choice-question","title":"MCQG-SRefine: Multiple Choice Question Generation and Evaluation with Iterative Self-Critique, Correction, and Comparison Feedback","date":"2024-10-17","arxiv_id":"2410.13191","n_code_links":1,"syntology":null},{"paper":"/paper/measuring-and-modifying-the-readability-of","slug":"measuring-and-modifying-the-readability-of","title":"Measuring and Modifying the Readability of English Texts with GPT-4","date":"2024-10-17","arxiv_id":"2410.14028","n_code_links":1,"syntology":null},{"paper":"/paper/sbi-rag-enhancing-math-word-problem-solving","slug":"sbi-rag-enhancing-math-word-problem-solving","title":"SBI-RAG: Enhancing Math Word Problem Solving for Students through Schema-Based Instruction and Retrieval-Augmented Generation","date":"2024-10-17","arxiv_id":"2410.13293","n_code_links":1,"syntology":null},{"paper":"/paper/towards-cross-cultural-machine-translation","slug":"towards-cross-cultural-machine-translation","title":"Towards Cross-Cultural Machine Translation with Retrieval-Augmented Generation from Multilingual Knowledge Graphs","date":"2024-10-17","arxiv_id":"2410.14057","n_code_links":0,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/at-rag-an-adaptive-rag-model-enhancing-query","slug":"at-rag-an-adaptive-rag-model-enhancing-query","title":"AT-RAG: An Adaptive RAG Model Enhancing Query Efficiency with Topic Filtering and Iterative Reasoning","date":"2024-10-16","arxiv_id":"2410.12886","n_code_links":1,"syntology":null},{"paper":null,"slug":"ccsbench-evaluating-compositional","title":"CCSBench: Evaluating Compositional Controllability in LLMs for Scientific Document Summarization","date":"2024-10-16","arxiv_id":"2410.12601","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-morphological-compositional","slug":"evaluating-morphological-compositional","title":"Evaluating Morphological Compositional Generalization in Large Language Models","date":"2024-10-16","arxiv_id":"2410.12656","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["selimfirat/bilkent-turkish-writings-dataset"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"identifying-task-groupings-for-multi-task","title":"Identifying Task Groupings for Multi-Task Learning Using Pointwise V-Usable Information","date":"2024-10-16","arxiv_id":"2410.12774","n_code_links":0,"syntology":null},{"paper":null,"slug":"mirror-a-novel-approach-for-the-automated","title":"MIRROR: A Novel Approach for the Automated Evaluation of Open-Ended Question Generation","date":"2024-10-16","arxiv_id":"2410.12893","n_code_links":0,"syntology":null},{"paper":"/paper/msc-sql-multi-sample-critiquing-small","slug":"msc-sql-multi-sample-critiquing-small","title":"MSc-SQL: Multi-Sample Critiquing Small Language Models For Text-To-SQL Translation","date":"2024-10-16","arxiv_id":"2410.12916","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["layer6ai-labs/msc-sql"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-a-scale-from-1-to-5-quantifying","title":"On A Scale From 1 to 5: Quantifying Hallucination in Faithfulness Evaluation","date":"2024-10-16","arxiv_id":"2410.12222","n_code_links":0,"syntology":null},{"paper":null,"slug":"table-llm-specialist-language-model","title":"Table-LLM-Specialist: Language Model Specialists for Tables using Iterative Generator-Validator Fine-tuning","date":"2024-10-16","arxiv_id":"2410.12164","n_code_links":0,"syntology":null},{"paper":"/paper/cognitive-overload-attack-prompt-injection","slug":"cognitive-overload-attack-prompt-injection","title":"Cognitive Overload Attack:Prompt Injection for Long Context","date":"2024-10-15","arxiv_id":"2410.11272","n_code_links":1,"syntology":null},{"paper":"/paper/de-jargonizing-science-for-journalists-with","slug":"de-jargonizing-science-for-journalists-with","title":"De-jargonizing Science for Journalists with GPT-4: A Pilot Study","date":"2024-10-15","arxiv_id":"2410.12069","n_code_links":1,"syntology":null},{"paper":"/paper/jigsaw-puzzles-splitting-harmful-questions-to","slug":"jigsaw-puzzles-splitting-harmful-questions-to","title":"Jigsaw Puzzles: Splitting Harmful Questions to Jailbreak Large Language Models","date":"2024-10-15","arxiv_id":"2410.11459","n_code_links":1,"syntology":null},{"paper":null,"slug":"selection-p-self-supervised-task-agnostic","title":"Selection-p: Self-Supervised Task-Agnostic Prompt Compression for Faithfulness and Transferability","date":"2024-10-15","arxiv_id":"2410.11786","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-realistic-evaluation-of-commit","title":"Towards Realistic Evaluation of Commit Message Generation by Matching Online and Offline Settings","date":"2024-10-15","arxiv_id":"2410.12046","n_code_links":0,"syntology":null},{"paper":null,"slug":"code-mixer-ya-nahi-novel-approaches-to","title":"Code-Mixer Ya Nahi: Novel Approaches to Measuring Multilingual LLMs' Code-Mixing Capabilities","date":"2024-10-14","arxiv_id":"2410.11079","n_code_links":0,"syntology":null},{"paper":null,"slug":"embedding-self-correction-as-an-inherent","title":"Embedding Self-Correction as an Inherent Ability in Large Language Models for Enhanced Mathematical Reasoning","date":"2024-10-14","arxiv_id":"2410.10735","n_code_links":0,"syntology":null},{"paper":"/paper/formalalign-automated-alignment-evaluation","slug":"formalalign-automated-alignment-evaluation","title":"FormalAlign: Automated Alignment Evaluation for Autoformalization","date":"2024-10-14","arxiv_id":"2410.10135","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["rookie-joe/formalalign"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gender-bias-of-llm-in-economics-an","title":"Gender Bias of LLM in Economics: An Existentialism Perspective","date":"2024-10-14","arxiv_id":"2410.19775","n_code_links":0,"syntology":null},{"paper":null,"slug":"generative-ai-and-its-impact-on-personalized","title":"Generative AI and Its Impact on Personalized Intelligent Tutoring Systems","date":"2024-10-14","arxiv_id":"2410.10650","n_code_links":0,"syntology":null},{"paper":null,"slug":"3ds-decomposed-difficulty-data-selection-s","title":"3DS: Decomposed Difficulty Data Selection's Case Study on LLM Medical Domain Adaptation","date":"2024-10-13","arxiv_id":"2410.10901","n_code_links":0,"syntology":null},{"paper":"/paper/easyjudge-an-easy-to-use-tool-for","slug":"easyjudge-an-easy-to-use-tool-for","title":"EasyJudge: an Easy-to-use Tool for Comprehensive Response Evaluation of LLMs","date":"2024-10-13","arxiv_id":"2410.09775","n_code_links":1,"syntology":null},{"paper":null,"slug":"empowering-dysarthric-speech-leveraging","title":"Empowering Dysarthric Speech: Leveraging Advanced LLMs for Accurate Speech Correction and Multimodal Emotion Analysis","date":"2024-10-13","arxiv_id":"2410.12867","n_code_links":0,"syntology":null},{"paper":"/paper/hardmath-a-benchmark-dataset-for-challenging","slug":"hardmath-a-benchmark-dataset-for-challenging","title":"HARDMath: A Benchmark Dataset for Challenging Problems in Applied Mathematics","date":"2024-10-13","arxiv_id":"2410.09988","n_code_links":1,"syntology":null},{"paper":null,"slug":"extended-japanese-commonsense-morality","title":"Extended Japanese Commonsense Morality Dataset with Masked Token and Label Enhancement","date":"2024-10-12","arxiv_id":"2410.09564","n_code_links":0,"syntology":null},{"paper":null,"slug":"gpton-generative-pre-trained-transformers","title":"GPTON: Generative Pre-trained Transformers enhanced with Ontology Narration for accurate annotation of biological data","date":"2024-10-12","arxiv_id":"2410.10899","n_code_links":0,"syntology":null},{"paper":"/paper/attngcg-enhancing-jailbreaking-attacks-on","slug":"attngcg-enhancing-jailbreaking-attacks-on","title":"AttnGCG: Enhancing Jailbreaking Attacks on LLMs with Attention Manipulation","date":"2024-10-11","arxiv_id":"2410.09040","n_code_links":1,"syntology":null},{"paper":"/paper/developing-a-pragmatic-benchmark-for","slug":"developing-a-pragmatic-benchmark-for","title":"Developing a Pragmatic Benchmark for Assessing Korean Legal Language Understanding in Large Language Models","date":"2024-10-11","arxiv_id":"2410.08731","n_code_links":1,"syntology":null},{"paper":null,"slug":"fine-tuning-in-house-large-language-models-to","title":"Fine-Tuning In-House Large Language Models to Infer Differential Diagnosis from Radiology Reports","date":"2024-10-11","arxiv_id":"2410.09234","n_code_links":0,"syntology":null},{"paper":null,"slug":"hypothesis-only-biases-in-large-language","title":"Hypothesis-only Biases in Large Language Model-Elicited Natural Language Inference","date":"2024-10-11","arxiv_id":"2410.08996","n_code_links":0,"syntology":null},{"paper":"/paper/jailjudge-a-comprehensive-jailbreak-judge","slug":"jailjudge-a-comprehensive-jailbreak-judge","title":"JAILJUDGE: A Comprehensive Jailbreak Judge Benchmark with Multi-Agent Enhanced Explanation Evaluation Framework","date":"2024-10-11","arxiv_id":"2410.12855","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"large-language-models-for-medical-osce","title":"Large Language Models for Medical OSCE Assessment: A Novel Approach to Transcript Analysis","date":"2024-10-11","arxiv_id":"2410.12858","n_code_links":0,"syntology":null},{"paper":"/paper/supercorrect-supervising-and-correcting","slug":"supercorrect-supervising-and-correcting","title":"SuperCorrect: Supervising and Correcting Language Models with Error-Driven Insights","date":"2024-10-11","arxiv_id":"2410.09008","n_code_links":2,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["yangling0818/supercorrect-llm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":"/paper/benchmarking-agentic-workflow-generation","slug":"benchmarking-agentic-workflow-generation","title":"Benchmarking Agentic Workflow Generation","date":"2024-10-10","arxiv_id":"2410.07869","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zjunlp/worfbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"diversity-of-thought-elicits-stronger","title":"Diversity of Thought Elicits Stronger Reasoning Capabilities in Multi-Agent Debate Frameworks","date":"2024-10-10","arxiv_id":"2410.12853","n_code_links":0,"syntology":null},{"paper":null,"slug":"plamo-100b-a-ground-up-language-model","title":"PLaMo-100B: A Ground-Up Language Model Designed for Japanese Proficiency","date":"2024-10-10","arxiv_id":"2410.07563","n_code_links":0,"syntology":null},{"paper":null,"slug":"prompt-engineering-a-schizophrenia-chatbot","title":"Prompt Engineering a Schizophrenia Chatbot: Utilizing a Multi-Agent Approach for Enhanced Compliance with Prompt Instructions","date":"2024-10-10","arxiv_id":"2410.12848","n_code_links":0,"syntology":null}],"record_sha256":"14f078cb68c4b7617030b11b79e9cf4bf74f7d8be5e845da12638264295f14ce","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}