{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/instruction-following/papers/7","list_of":"/task/instruction-following","task":"Instruction Following","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":12,"rows_per_page":100,"rows":[601,700],"of":1135,"counts":{"archive_papers_tagged":1135,"with_a_code_link":609,"where_syntology_ran_a_sample":311,"not_listed_spam_title":0,"listed":1135,"listed_where_code_ran":311,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":255,"every_run_a_failure_of_syntologys_instrument":56,"listed_with_a_run_with_no_instrument_failure":255,"listed_every_run_a_failure_of_syntologys_instrument":56,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/instruction-following","prev":"/task/instruction-following/papers/6","next":"/task/instruction-following/papers/8","papers":[{"url":"/paper/babyai-towards-grounded-language-learning","slug":"babyai-towards-grounded-language-learning","title":"Zero-Shot Compositional Policy Learning via Language Grounding","date":"2020-04-15","arxiv_id":"2004.07200","repositories_listed":1,"syntology":null},{"url":"/paper/automated-curriculum-generation-for-policy","slug":"automated-curriculum-generation-for-policy","title":"Automated curriculum generation for Policy Gradients from Demonstrations","date":"2019-12-01","arxiv_id":"1912.00444","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-map-natural-language-instructions","slug":"learning-to-map-natural-language-instructions","title":"Learning to Map Natural Language Instructions to Physical Quadcopter Control using Simulated Flight","date":"2019-10-21","arxiv_id":"1910.09664","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-map-natural-language-instructions#ran","syntology_url":"https://syntology.ai/paper/1910.09664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09664"}},"official":{"repos":["lil-lab/drif"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-learning-environment-representations-for","slug":"pre-learning-environment-representations-for","title":"Pre-Learning Environment Representations for Data-Efficient Neural Instruction Following","date":"2019-07-23","arxiv_id":"1907.09671","repositories_listed":1,"syntology":null},{"url":"/paper/chasing-ghosts-instruction-following-as","slug":"chasing-ghosts-instruction-following-as","title":"Chasing Ghosts: Instruction Following as Bayesian State Tracking","date":"2019-07-03","arxiv_id":"1907.02022","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chasing-ghosts-instruction-following-as#ran","syntology_url":"https://syntology.ai/paper/1907.02022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1907.02022"}},"official":{"repos":["batra-mlp-lab/vln-chasing-ghosts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/guiding-policies-with-language-via-meta","slug":"guiding-policies-with-language-via-meta","title":"Guiding Policies with Language via Meta-Learning","date":"2018-11-19","arxiv_id":"1811.07882","repositories_listed":1,"syntology":null},{"url":"/paper/mapping-navigation-instructions-to-continuous","slug":"mapping-navigation-instructions-to-continuous","title":"Mapping Navigation Instructions to Continuous Control Actions with Position-Visitation Prediction","date":"2018-11-10","arxiv_id":"1811.04179","repositories_listed":1,"syntology":null},{"url":"/paper/following-high-level-navigation-instructions","slug":"following-high-level-navigation-instructions","title":"Following High-level Navigation Instructions on a Simulated Quadcopter with Imitation Learning","date":"2018-05-31","arxiv_id":"1806.00047","repositories_listed":1,"syntology":null},{"url":"/paper/alignment-based-compositional-semantics-for","slug":"alignment-based-compositional-semantics-for","title":"Alignment-based compositional semantics for instruction following","date":"2015-08-26","arxiv_id":"1508.06491","repositories_listed":1,"syntology":null},{"url":null,"slug":"anycap-project-a-unified-framework-dataset","title":"AnyCap Project: A Unified Framework, Dataset, and Benchmark for Controllable Omni-modal Captioning","date":"2025-07-17","arxiv_id":"2507.12841","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-many-instructions-can-llms-follow-at-once","title":"How Many Instructions Can LLMs Follow at Once?","date":"2025-07-15","arxiv_id":"2507.11538","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-multimodal-software-developer","title":"Multilingual Multimodal Software Developer for Code Generation","date":"2025-07-11","arxiv_id":"2507.08719","repositories_listed":0,"syntology":null},{"url":null,"slug":"tuneshield-mitigating-toxicity-in","title":"TuneShield: Mitigating Toxicity in Conversational AI while Fine-tuning on Untrusted Data","date":"2025-07-08","arxiv_id":"2507.05660","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-offline-and-online-reinforcement","title":"Bridging Offline and Online Reinforcement Learning for LLMs","date":"2025-06-26","arxiv_id":"2506.21495","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-lingual-functional-evaluation-for-large","title":"Multi-lingual Functional Evaluation for Large Language Models","date":"2025-06-25","arxiv_id":"2506.20793","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-instruction-following-policies","title":"Learning Instruction-Following Policies through Open-Ended Instruction Relabeling with Large Language Models","date":"2025-06-24","arxiv_id":"2506.20061","repositories_listed":0,"syntology":null},{"url":null,"slug":"jarvisart-liberating-human-artistic","title":"JarvisArt: Liberating Human Artistic Creativity via an Intelligent Photo Retouching Agent","date":"2025-06-21","arxiv_id":"2506.17612","repositories_listed":0,"syntology":null},{"url":null,"slug":"treasure-hunt-real-time-targeting-of-the-long","title":"Treasure Hunt: Real-time Targeting of the Long Tail using Training-Time Markers","date":"2025-06-17","arxiv_id":"2506.14702","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-following-by-boosting-attention","title":"Instruction Following by Boosting Attention of Large Language Models","date":"2025-06-16","arxiv_id":"2506.13734","repositories_listed":0,"syntology":null},{"url":null,"slug":"leverb-humanoid-whole-body-control-with","title":"LeVERB: Humanoid Whole-Body Control with Latent Vision-Language Instruction","date":"2025-06-16","arxiv_id":"2506.13751","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixture-of-weight-shared-heterogeneous-group","title":"Mixture of Weight-shared Heterogeneous Group Attention Experts for Dynamic Token-wise KV Optimization","date":"2025-06-16","arxiv_id":"2506.13541","repositories_listed":0,"syntology":null},{"url":null,"slug":"cmi-bench-a-comprehensive-benchmark-for","title":"CMI-Bench: A Comprehensive Benchmark for Evaluating Music Instruction Following","date":"2025-06-14","arxiv_id":"2506.12285","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-r5-multimodal-reasoning-enhanced-reranker","title":"MM-R5: MultiModal Reasoning-Enhanced ReRanker via Reinforcement Learning for Document Retrieval","date":"2025-06-14","arxiv_id":"2506.12364","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10312","title":"AC/DC: LLM-based Audio Comprehension via Dialogue Continuation","date":"2025-06-12","arxiv_id":"2506.10312","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-search-from-fundamentals-to","title":"Conversational Search: From Fundamentals to Frontiers in the LLM Era","date":"2025-06-12","arxiv_id":"2506.10635","repositories_listed":0,"syntology":null},{"url":null,"slug":"halloc-token-level-localization-of-1","title":"HalLoc: Token-level Localization of Hallucinations for Vision Language Models","date":"2025-06-12","arxiv_id":"2506.10286","repositories_listed":0,"syntology":null},{"url":null,"slug":"magistral","title":"Magistral","date":"2025-06-12","arxiv_id":"2506.10910","repositories_listed":0,"syntology":null},{"url":null,"slug":"alzheimer-s-dementia-detection-using","title":"Alzheimer's Dementia Detection Using Perplexity from Paired Large Language Models","date":"2025-06-11","arxiv_id":"2506.09315","repositories_listed":0,"syntology":null},{"url":null,"slug":"llava-c-continual-improved-visual-instruction","title":"LLaVA-c: Continual Improved Visual Instruction Tuning","date":"2025-06-10","arxiv_id":"2506.08666","repositories_listed":0,"syntology":null},{"url":null,"slug":"rhealthtwin-towards-responsible-and","title":"RHealthTwin: Towards Responsible and Multimodal Digital Twins for Personalized Well-being","date":"2025-06-10","arxiv_id":"2506.08486","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-text-images-and-3d-structure-token","title":"Aligning Text, Images, and 3D Structure Token-by-Token","date":"2025-06-09","arxiv_id":"2506.08002","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-unlearning-via-low-rank-refusal-vector","title":"Video Unlearning via Low-Rank Refusal Vector","date":"2025-06-09","arxiv_id":"2506.07891","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-aware-large-language-models-as-judges","title":"Audio-Aware Large Language Models as Judges for Speaking Styles","date":"2025-06-06","arxiv_id":"2506.05984","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-mechanism-of-reasoning-pattern","title":"On the Mechanism of Reasoning Pattern Selection in Reinforcement Learning for Language Models","date":"2025-06-05","arxiv_id":"2506.04695","repositories_listed":0,"syntology":null},{"url":null,"slug":"relic-evaluating-compositional-instruction","title":"RELIC: Evaluating Compositional Instruction Following via Language Recognition","date":"2025-06-05","arxiv_id":"2506.05205","repositories_listed":0,"syntology":null},{"url":null,"slug":"seededit-3-0-fast-and-high-quality-generative","title":"SeedEdit 3.0: Fast and High-Quality Generative Image Editing","date":"2025-06-05","arxiv_id":"2506.05083","repositories_listed":0,"syntology":null},{"url":null,"slug":"unleashing-hour-scale-video-training-for-long","title":"Unleashing Hour-Scale Video Training for Long Video-Language Understanding","date":"2025-06-05","arxiv_id":"2506.05332","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-anti-backdoor-instruction-tuning-in","title":"Robust Anti-Backdoor Instruction Tuning in LVLMs","date":"2025-06-04","arxiv_id":"2506.05401","repositories_listed":0,"syntology":null},{"url":null,"slug":"master-enhancing-large-language-model-via","title":"MASTER: Enhancing Large Language Model via Multi-Agent Simulated Teaching","date":"2025-06-03","arxiv_id":"2506.02689","repositories_listed":0,"syntology":null},{"url":null,"slug":"moda-modulation-adapter-for-fine-grained","title":"MoDA: Modulation Adapter for Fine-Grained Visual Grounding in Instructional MLLMs","date":"2025-06-02","arxiv_id":"2506.01850","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiif-bench-how-does-your-t2i-model-follow","title":"TIIF-Bench: How Does Your T2I Model Follow Your Instructions?","date":"2025-06-02","arxiv_id":"2506.02161","repositories_listed":0,"syntology":null},{"url":null,"slug":"persianmedqa-language-centric-evaluation-of","title":"PersianMedQA: Language-Centric Evaluation of LLMs in the Persian Medical Domain","date":"2025-05-30","arxiv_id":"2506.00250","repositories_listed":0,"syntology":null},{"url":null,"slug":"arc-argument-representation-and-coverage","title":"ARC: Argument Representation and Coverage Analysis for Zero-Shot Long Document Summarization with Instruction Following LLMs","date":"2025-05-29","arxiv_id":"2505.23654","repositories_listed":0,"syntology":null},{"url":null,"slug":"chartmind-a-comprehensive-benchmark-for","title":"ChartMind: A Comprehensive Benchmark for Complex Real-world Multimodal Chart Question Answering","date":"2025-05-29","arxiv_id":"2505.23242","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-information-an-information","title":"Differential Information: An Information-Theoretic Perspective on Preference Optimization","date":"2025-05-29","arxiv_id":"2505.23761","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-detoxification-safeguarding-general","title":"Adaptive Detoxification: Safeguarding General Capabilities of LLMs through Toxicity-Aware Knowledge Editing","date":"2025-05-28","arxiv_id":"2505.22298","repositories_listed":0,"syntology":null},{"url":null,"slug":"lamdagent-an-autonomous-framework-for-post","title":"LaMDAgent: An Autonomous Framework for Post-Training Pipeline Optimization via LLM Agents","date":"2025-05-28","arxiv_id":"2505.21963","repositories_listed":0,"syntology":null},{"url":null,"slug":"partinstruct-part-level-instruction-following","title":"PartInstruct: Part-level Instruction Following for Fine-grained Robot Manipulation","date":"2025-05-27","arxiv_id":"2505.21652","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-robustness-of-large-audio-language","title":"Evaluating Robustness of Large Audio Language Models to Audio Injection: An Empirical Study","date":"2025-05-26","arxiv_id":"2505.19598","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-alignment-to-advancement-bootstrapping","title":"From Alignment to Advancement: Bootstrapping Audio-Language Alignment with Synthetic Data","date":"2025-05-26","arxiv_id":"2505.20166","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-importance-sampling-to-detach","title":"Leveraging Importance Sampling to Detach Alignment Modules from Large Language Models","date":"2025-05-26","arxiv_id":"2505.19700","repositories_listed":0,"syntology":null},{"url":null,"slug":"stylear-customizing-multimodal-autoregressive","title":"StyleAR: Customizing Multimodal Autoregressive Model for Style-Aligned Text-to-Image Generation","date":"2025-05-26","arxiv_id":"2505.19874","repositories_listed":0,"syntology":null},{"url":null,"slug":"recast-strengthening-llms-complex-instruction","title":"RECAST: Strengthening LLMs' Complex Instruction Following with Constraint-Verifiable Data","date":"2025-05-25","arxiv_id":"2505.19030","repositories_listed":0,"syntology":null},{"url":null,"slug":"midb-multilingual-instruction-data-booster","title":"MIDB: Multilingual Instruction Data Booster for Enhancing Multilingual Instruction Synthesis","date":"2025-05-23","arxiv_id":"2505.17671","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-watermarks-for-large-language","title":"In-Context Watermarks for Large Language Models","date":"2025-05-22","arxiv_id":"2505.16934","repositories_listed":0,"syntology":null},{"url":null,"slug":"maniplvm-r1-reinforcement-learning-for","title":"ManipLVM-R1: Reinforcement Learning for Reasoning in Embodied Manipulation with Large Vision-Language Models","date":"2025-05-22","arxiv_id":"2505.16517","repositories_listed":0,"syntology":null},{"url":"/paper/sparse-activation-editing-for-reliable","slug":"sparse-activation-editing-for-reliable","title":"Sparse Activation Editing for Reliable Instruction Following in Narratives","date":"2025-05-22","arxiv_id":"2505.16505","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-activation-editing-for-reliable#ran","syntology_url":"https://syntology.ai/paper/2505.16505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16505"}},"official":null}},{"url":null,"slug":"todi-token-wise-distillation-via-fine-grained","title":"ToDi: Token-wise Distillation via Fine-Grained Divergence Control","date":"2025-05-22","arxiv_id":"2505.16297","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-vs-autoregressive-language-models-a","title":"Diffusion vs. Autoregressive Language Models: A Text Embedding Perspective","date":"2025-05-21","arxiv_id":"2505.15045","repositories_listed":0,"syntology":null},{"url":null,"slug":"flowkv-enhancing-multi-turn-conversational","title":"FlowKV: Enhancing Multi-Turn Conversational Coherence in LLMs via Isolated Key-Value Cache Management","date":"2025-05-21","arxiv_id":"2505.15347","repositories_listed":0,"syntology":null},{"url":null,"slug":"hunyuan-turbos-advancing-large-language","title":"Hunyuan-TurboS: Advancing Large Language Models through Mamba-Transformer Synergy and Adaptive Chain-of-Thought","date":"2025-05-21","arxiv_id":"2505.15431","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-flashback-adaptation-for-forgetting","title":"Joint Flashback Adaptation for Forgetting-Resistant Instruction Tuning","date":"2025-05-21","arxiv_id":"2505.15467","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinkless-a-training-free-inference-efficient","title":"ThinkLess: A Training-Free Inference-Efficient Method for Reducing Reasoning Redundancy","date":"2025-05-21","arxiv_id":"2505.15684","repositories_listed":0,"syntology":null},{"url":null,"slug":"decif-improving-instruction-following-through","title":"DecIF: Improving Instruction-Following through Meta-Decomposition","date":"2025-05-20","arxiv_id":"2505.13990","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-of-vlm-for-soccer-video","title":"Domain Adaptation of VLM for Soccer Video Understanding","date":"2025-05-20","arxiv_id":"2505.13860","repositories_listed":0,"syntology":null},{"url":null,"slug":"ground-v-teaching-vlms-to-ground-complex","title":"Ground-V: Teaching VLMs to Ground Complex Instructions in Pixels","date":"2025-05-20","arxiv_id":"2505.13788","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-experts-are-all-you-need-for-steering","title":"Two Experts Are All You Need for Steering Thinking: Reinforcing Cognitive Effort in MoE Reasoning Models Without Additional Training","date":"2025-05-20","arxiv_id":"2505.14681","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-head-gating-a-framework-for","title":"Causal Head Gating: A Framework for Interpreting Roles of Attention Heads in Transformers","date":"2025-05-19","arxiv_id":"2505.13737","repositories_listed":0,"syntology":null},{"url":null,"slug":"kit-s-offline-speech-translation-and","title":"KIT's Offline Speech Translation and Instruction Following Submission for IWSLT 2025","date":"2025-05-19","arxiv_id":"2505.13036","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-level-aware-preference-learning","title":"Multi-Level Aware Preference Learning: Enhancing RLHF for Complex Multi-Instruction Tasks","date":"2025-05-19","arxiv_id":"2505.12845","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-predictive-modeling-for-llm","title":"Rethinking Predictive Modeling for LLM Routing: When Simple kNN Beats Complex Learned Routers","date":"2025-05-19","arxiv_id":"2505.12601","repositories_listed":0,"syntology":null},{"url":null,"slug":"compbench-benchmarking-complex-instruction","title":"CompBench: Benchmarking Complex Instruction-guided Image Editing","date":"2025-05-18","arxiv_id":"2505.12200","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-complex-instruction-following-for","title":"Enhancing Complex Instruction Following for Large Language Models with Mixture-of-Contexts Fine-tuning","date":"2025-05-17","arxiv_id":"2505.11922","repositories_listed":0,"syntology":null},{"url":"/paper/2505-11368","slug":"2505-11368","title":"GuideBench: Benchmarking Domain-Oriented Guideline Following for LLM Agents","date":"2025-05-16","arxiv_id":"2505.11368","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/2505-11368#ran","syntology_url":"https://syntology.ai/paper/2505.11368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11368"}},"official":null}},{"url":null,"slug":"2505-11423","title":"When Thinking Fails: The Pitfalls of Reasoning for Instruction-Following in LLMs","date":"2025-05-16","arxiv_id":"2505.11423","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11475","title":"HelpSteer3-Preference: Open Human-Annotated Preference Data across Diverse Tasks and Languages","date":"2025-05-16","arxiv_id":"2505.11475","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-the-alpha-jungle-an-llm-powered","title":"Navigating the Alpha Jungle: An LLM-Powered MCTS Framework for Formulaic Factor Mining","date":"2025-05-16","arxiv_id":"2505.11122","repositories_listed":0,"syntology":null},{"url":null,"slug":"unieval-unified-holistic-evaluation-for","title":"UniEval: Unified Holistic Evaluation for Unified Multimodal Understanding and Generation","date":"2025-05-15","arxiv_id":"2505.10483","repositories_listed":0,"syntology":null},{"url":null,"slug":"tests-as-prompt-a-test-driven-development","title":"Tests as Prompt: A Test-Driven-Development Benchmark for LLM Code Generation","date":"2025-05-13","arxiv_id":"2505.09027","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-telecom-specific-llm-tslam-mini","title":"Efficient Telecom Specific LLM: TSLAM-Mini with QLoRA and Digital Twin Data","date":"2025-05-10","arxiv_id":"2505.07877","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-robustness-to-spurious-correlations","title":"Assessing Robustness to Spurious Correlations in Post-Training Language Models","date":"2025-05-09","arxiv_id":"2505.05704","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2vtextbench-a-human-evaluation-benchmark-for","title":"T2VTextBench: A Human Evaluation Benchmark for Textual Control in Video Generation Models","date":"2025-05-08","arxiv_id":"2505.04946","repositories_listed":0,"syntology":null},{"url":null,"slug":"incentivizing-inclusive-contributions-in","title":"Incentivizing Inclusive Contributions in Model Sharing Markets","date":"2025-05-05","arxiv_id":"2505.02462","repositories_listed":0,"syntology":null},{"url":null,"slug":"pipa-a-unified-evaluation-protocol-for","title":"PIPA: A Unified Evaluation Protocol for Diagnosing Interactive Planning Agents","date":"2025-05-02","arxiv_id":"2505.01592","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2vphysbench-a-first-principles-benchmark-for","title":"T2VPhysBench: A First-Principles Benchmark for Physical Consistency in Text-to-Video Generation","date":"2025-05-01","arxiv_id":"2505.00337","repositories_listed":0,"syntology":null},{"url":null,"slug":"meeseeks-an-iterative-benchmark-evaluating","title":"Ask, Fail, Repeat: Meeseeks, an Iterative Feedback Benchmark for LLMs' Multi-turn Instruction-Following Ability","date":"2025-04-30","arxiv_id":"2504.21625","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-vln-end-to-end-vision-language-guided","title":"UAV-VLN: End-to-End Vision Language guided Navigation for UAVs","date":"2025-04-30","arxiv_id":"2504.21432","repositories_listed":0,"syntology":null},{"url":null,"slug":"cacheprune-neural-based-attribution-defense","title":"CachePrune: Neural-Based Attribution Defense Against Indirect Prompt Injection Attacks","date":"2025-04-29","arxiv_id":"2504.21228","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-the-modality-barrier-universal","title":"Breaking the Modality Barrier: Universal Embedding Learning with Multimodal LLMs","date":"2025-04-24","arxiv_id":"2504.17432","repositories_listed":0,"syntology":null},{"url":"/paper/case-study-fine-tuning-small-language-models","slug":"case-study-fine-tuning-small-language-models","title":"Case Study: Fine-tuning Small Language Models for Accurate and Private CWE Detection in Python Code","date":"2025-04-23","arxiv_id":"2504.16584","repositories_listed":0,"syntology":null},{"url":null,"slug":"manipdreamer-boosting-robotic-manipulation","title":"ManipDreamer: Boosting Robotic Manipulation World Model with Action Tree and Visual Guidance","date":"2025-04-23","arxiv_id":"2504.16464","repositories_listed":0,"syntology":null},{"url":null,"slug":"param-d-for-direct-weight-mixing-post-train","title":"Param$Δ$ for Direct Weight Mixing: Post-Train Large Language Model at Zero Cost","date":"2025-04-23","arxiv_id":"2504.21023","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilqwen2-5-industrial-practices-of","title":"DistilQwen2.5: Industrial Practices of Training Distilled Open Lightweight Language Models","date":"2025-04-21","arxiv_id":"2504.15027","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-instruct-models-for-free-a-study-on","title":"Improving Instruct Models for Free: A Study on Partial Adaptation","date":"2025-04-15","arxiv_id":"2504.11626","repositories_listed":0,"syntology":null},{"url":null,"slug":"sift-50m-a-large-scale-multilingual-dataset","title":"SIFT-50M: A Large-Scale Multilingual Dataset for Speech Instruction Fine-Tuning","date":"2025-04-12","arxiv_id":"2504.09081","repositories_listed":0,"syntology":null},{"url":null,"slug":"capybara-omni-an-efficient-paradigm-for","title":"Capybara-OMNI: An Efficient Paradigm for Building Omni-Modal Language Models","date":"2025-04-10","arxiv_id":"2504.12315","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoexpert-augmented-llm-for-temporal","title":"VideoExpert: Augmented LLM for Temporal-Sensitive Video Understanding","date":"2025-04-10","arxiv_id":"2504.07519","repositories_listed":0,"syntology":null},{"url":null,"slug":"holistic-capability-preservation-towards","title":"Holistic Capability Preservation: Towards Compact Yet Comprehensive Reasoning Models","date":"2025-04-09","arxiv_id":"2504.07158","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-fantastic-experts-in-moes-a-unified","title":"Finding Fantastic Experts in MoEs: A Unified Study for Expert Dropping Strategies and Observations","date":"2025-04-08","arxiv_id":"2504.05586","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-128k-to-4m-efficient-training-of-ultra","title":"From 128K to 4M: Efficient Training of Ultra-Long Context Large Language Models","date":"2025-04-08","arxiv_id":"2504.06214","repositories_listed":0,"syntology":null}],"record_sha256":"9e4363014dbc50bedc9fa3f827bf5b97aa3358a2c1ddec3ae77978a5402295c2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}