{"url":"/task/large-language-model","name":"Large Language Model","slug":"large-language-model","description_markdown":null,"categories":[{"name":"Knowledge Base","url":"/area/knowledge-base"},{"name":"Methodology","url":"/area/methodology"},{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":6097,"papers_with_code":2250,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":10,"subtasks":3,"parent_tasks":2},"benchmarks":[],"datasets":[{"url":"/dataset/ifeval","name":"IFEval","full_name":"Instruction Following Evaluation Datset","num_papers_in_archive":180},{"url":"/dataset/medconceptsqa","name":"MedConceptsQA","full_name":"","num_papers_in_archive":13},{"url":"/dataset/triviahg","name":"TriviaHG","full_name":"","num_papers_in_archive":3},{"url":"/dataset/human-simulacra","name":"Human Simulacra","full_name":"","num_papers_in_archive":2},{"url":"/dataset/diaforge-utc-r-0725","name":"diaforge-utc-r-0725","full_name":"DiaFORGE UTC: Unified Tool-Calling Conversations Dataset","num_papers_in_archive":1},{"url":"/dataset/glotsparse","name":"GlotSparse","full_name":"","num_papers_in_archive":1},{"url":"/dataset/hixstest","name":"HiXSTest","full_name":"Hindi XSTest","num_papers_in_archive":1},{"url":"/dataset/rpeval","name":"RPEval","full_name":"Role-Playing Evaluation Dataset","num_papers_in_archive":1},{"url":"/dataset/sgxstest","name":"SGXSTest","full_name":"Singapore XSTest","num_papers_in_archive":1},{"url":"/dataset/vmd","name":"VMD","full_name":"Virtual Moderation Dataset","num_papers_in_archive":1}],"subtasks":[{"url":"/task/ai-agent","name":"AI Agent"},{"url":"/task/knowledge-graphs","name":"Knowledge Graphs"},{"url":"/task/rag","name":"RAG"}],"parent_tasks":[{"url":"/task/inductive-knowledge-graph-completion","name":"Inductive knowledge graph completion"},{"url":"/task/knowledge-graph-completion","name":"Knowledge Graph Completion"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":2250,"tagged_in_all":6097,"items":[{"url":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","date":"2023-06-09","arxiv_id":"2306.05685","repositories_listed":11,"syntology":{"n":12,"n_ran":0,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/generative-agents-interactive-simulacra-of","title":"Generative Agents: Interactive Simulacra of Human Behavior","date":"2023-04-07","arxiv_id":"2304.03442","repositories_listed":8,"syntology":{"n":12,"n_ran":5,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/a-conversational-paradigm-for-program","title":"CodeGen: An Open Large Language Model for Code with Multi-Turn Program Synthesis","date":"2022-03-25","arxiv_id":"2203.13474","repositories_listed":8,"syntology":{"n":10,"n_ran":10,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/hybridflow-a-flexible-and-efficient-rlhf","title":"HybridFlow: A Flexible and Efficient RLHF Framework","date":"2024-09-28","arxiv_id":"2409.19256","repositories_listed":7,"syntology":{"n":16,"n_ran":9,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/2309-06180","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","date":"2023-09-12","arxiv_id":"2309.06180","repositories_listed":7,"syntology":{"n":36,"n_ran":25,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/an-example-of-evolutionary-computation-large","title":"Evolution of Heuristics: Towards Efficient Automatic Algorithm Design Using Large Language Model","date":"2024-01-04","arxiv_id":"2401.02051","repositories_listed":6,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","repositories_listed":6,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/deepspeed-ulysses-system-optimizations-for","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","date":"2023-09-25","arxiv_id":"2309.14509","repositories_listed":6,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/minigpt-4-enhancing-vision-language","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","date":"2023-04-20","arxiv_id":"2304.10592","repositories_listed":6,"syntology":null},{"url":"/paper/xiyan-sql-a-multi-generator-ensemble","title":"A Preview of XiYan-SQL: A Multi-Generator Ensemble Framework for Text-to-SQL","date":"2024-11-13","arxiv_id":"2411.08599","repositories_listed":5,"syntology":null},{"url":"/paper/llm-as-os-llmao-agents-as-apps-envisioning","title":"LLM as OS, Agents as Apps: Envisioning AIOS, Agents and the AIOS-Agent Ecosystem","date":"2023-12-06","arxiv_id":"2312.03815","repositories_listed":5,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/point-bind-point-llm-aligning-point-cloud","title":"Point-Bind & Point-LLM: Aligning Point Cloud with Multi-modality for 3D Understanding, Generation, and Instruction Following","date":"2023-09-01","arxiv_id":"2309.00615","repositories_listed":5,"syntology":{"n":20,"n_ran":13,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/baize-an-open-source-chat-model-with","title":"Baize: An Open-Source Chat Model with Parameter-Efficient Tuning on Self-Chat Data","date":"2023-04-03","arxiv_id":"2304.01196","repositories_listed":5,"syntology":{"n":9,"n_ran":1,"n_unverified":8,"n_pointer_only":1}},{"url":"/paper/accelerating-large-language-model-decoding","title":"Accelerating Large Language Model Decoding with Speculative Sampling","date":"2023-02-02","arxiv_id":"2302.01318","repositories_listed":5,"syntology":null},{"url":"/paper/muse-text-to-image-generation-via-masked","title":"Muse: Text-To-Image Generation via Masked Generative Transformers","date":"2023-01-02","arxiv_id":"2301.00704","repositories_listed":5,"syntology":{"n":21,"n_ran":18,"n_unverified":3,"n_pointer_only":9}},{"url":"/paper/a-mem-agentic-memory-for-llm-agents","title":"A-MEM: Agentic Memory for LLM Agents","date":"2025-02-17","arxiv_id":"2502.12110","repositories_listed":4,"syntology":{"n":10,"n_ran":6,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/scaling-synthetic-data-creation-with","title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","date":"2024-06-28","arxiv_id":"2406.20094","repositories_listed":4,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/ranking-manipulation-for-conversational","title":"Ranking Manipulation for Conversational Search Engines","date":"2024-06-05","arxiv_id":"2406.03589","repositories_listed":4,"syntology":{"n":20,"n_ran":12,"n_unverified":8,"n_pointer_only":20}},{"url":"/paper/qserve-w4a8kv4-quantization-and-system-co","title":"QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving","date":"2024-05-07","arxiv_id":"2405.04532","repositories_listed":4,"syntology":null},{"url":"/paper/tower-an-open-multilingual-large-language","title":"Tower: An Open Multilingual Large Language Model for Translation-Related Tasks","date":"2024-02-27","arxiv_id":"2402.17733","repositories_listed":4,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":7}},{"url":"/paper/x-lora-mixture-of-low-rank-adapter-experts-a","title":"X-LoRA: Mixture of Low-Rank Adapter Experts, a Flexible Framework for Large Language Models with Applications in Protein Mechanics and Molecular Design","date":"2024-02-11","arxiv_id":"2402.07148","repositories_listed":4,"syntology":null},{"url":"/paper/algorithm-evolution-using-large-language","title":"Algorithm Evolution Using Large Language Model","date":"2023-11-26","arxiv_id":"2311.15249","repositories_listed":4,"syntology":null},{"url":"/paper/nlp-evaluation-in-trouble-on-the-need-to","title":"NLP Evaluation in trouble: On the Need to Measure LLM Data Contamination for each Benchmark","date":"2023-10-27","arxiv_id":"2310.18018","repositories_listed":4,"syntology":null},{"url":"/paper/llemma-an-open-language-model-for-mathematics","title":"Llemma: An Open Language Model For Mathematics","date":"2023-10-16","arxiv_id":"2310.10631","repositories_listed":4,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/how-is-chatgpt-s-behavior-changing-over-time","title":"How is ChatGPT's behavior changing over time?","date":"2023-07-18","arxiv_id":"2307.09009","repositories_listed":4,"syntology":{"n":6,"n_ran":2,"n_unverified":4,"n_pointer_only":5}},{"url":"/paper/hyenadna-long-range-genomic-sequence-modeling","title":"HyenaDNA: Long-Range Genomic Sequence Modeling at Single Nucleotide Resolution","date":"2023-06-27","arxiv_id":"2306.15794","repositories_listed":4,"syntology":{"n":28,"n_ran":17,"n_unverified":11,"n_pointer_only":1}},{"url":"/paper/mme-a-comprehensive-evaluation-benchmark-for","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","date":"2023-06-23","arxiv_id":"2306.13394","repositories_listed":4,"syntology":null},{"url":"/paper/deeptextmark-deep-learning-based-text","title":"DeepTextMark: A Deep Learning-Driven Text Watermarking Approach for Identifying Large Language Model Generated Text","date":"2023-05-09","arxiv_id":"2305.05773","repositories_listed":4,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/protoformer-embedding-prototypes-for-1","title":"Protoformer: Embedding Prototypes for Transformers","date":"2022-06-25","arxiv_id":"2206.12710","repositories_listed":4,"syntology":null},{"url":"/paper/fast-transformer-decoding-one-write-head-is","title":"Fast Transformer Decoding: One Write-Head is All You Need","date":"2019-11-06","arxiv_id":"1911.02150","repositories_listed":4,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}}],"syntology_records":21,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}