{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/sft/papers/2","list_of":"/method/sft","method":"SFT","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":2,"pages_in_order":5,"rows_per_page":100,"rows":[101,200],"of":415,"counts":{"archive_papers_tagged":415,"with_a_code_link":204,"where_syntology_ran_a_sample":103,"not_listed_spam_title":0,"listed":415,"listed_where_code_ran":103,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":86,"every_run_a_failure_of_syntologys_instrument":17,"listed_with_a_run_with_no_instrument_failure":86,"listed_every_run_a_failure_of_syntologys_instrument":17,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/sft","prev":"/method/sft","next":"/method/sft/papers/3","papers":[{"paper":"/paper/embodied-r-collaborative-framework-for","slug":"embodied-r-collaborative-framework-for","title":"Embodied-R: Collaborative Framework for Activating Embodied Spatial Reasoning in Foundation Models via Reinforcement Learning","date":"2025-04-17","arxiv_id":"2504.12680","n_code_links":1,"syntology":null},{"paper":"/paper/skyreels-v2-infinite-length-film-generative","slug":"skyreels-v2-infinite-length-film-generative","title":"SkyReels-V2: Infinite-length Film Generative Model","date":"2025-04-17","arxiv_id":"2504.13074","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-pre-training-indicators-reliably-predict","title":"Can Pre-training Indicators Reliably Predict Fine-tuning Outcomes of LLMs?","date":"2025-04-16","arxiv_id":"2504.12491","n_code_links":0,"syntology":null},{"paper":"/paper/climbing-the-ladder-of-reasoning-what-llms","slug":"climbing-the-ladder-of-reasoning-what-llms","title":"Climbing the Ladder of Reasoning: What LLMs Can-and Still Can't-Solve after SFT?","date":"2025-04-16","arxiv_id":"2504.11741","n_code_links":1,"syntology":null},{"paper":null,"slug":"d1-scaling-reasoning-in-diffusion-large","title":"d1: Scaling Reasoning in Diffusion Large Language Models via Reinforcement Learning","date":"2025-04-16","arxiv_id":"2504.12216","n_code_links":0,"syntology":null},{"paper":null,"slug":"evaluating-the-diversity-and-quality-of-llm","title":"Evaluating the Diversity and Quality of LLM Generated Content","date":"2025-04-16","arxiv_id":"2504.12522","n_code_links":0,"syntology":null},{"paper":"/paper/toolrl-reward-is-all-tool-learning-needs","slug":"toolrl-reward-is-all-tool-learning-needs","title":"ToolRL: Reward is All Tool Learning Needs","date":"2025-04-16","arxiv_id":"2504.13958","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qiancheng0/toolrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"seedream-3-0-technical-report","title":"Seedream 3.0 Technical Report","date":"2025-04-15","arxiv_id":"2504.11346","n_code_links":0,"syntology":null},{"paper":null,"slug":"weight-ensembling-improves-reasoning-in","title":"Weight Ensembling Improves Reasoning in Language Models","date":"2025-04-14","arxiv_id":"2504.10478","n_code_links":0,"syntology":null},{"paper":"/paper/llms-can-achieve-high-quality-simultaneous","slug":"llms-can-achieve-high-quality-simultaneous","title":"LLMs Can Achieve High-quality Simultaneous Machine Translation as Efficiently as Offline","date":"2025-04-13","arxiv_id":"2504.09570","n_code_links":1,"syntology":null},{"paper":null,"slug":"pathvlm-r1-a-reinforcement-learning-driven","title":"PathVLM-R1: A Reinforcement Learning-Driven Reasoning Model for Pathology Visual-Language Tasks","date":"2025-04-12","arxiv_id":"2504.09258","n_code_links":0,"syntology":null},{"paper":"/paper/mm-ifengine-towards-multimodal-instruction","slug":"mm-ifengine-towards-multimodal-instruction","title":"MM-IFEngine: Towards Multimodal Instruction Following","date":"2025-04-10","arxiv_id":"2504.07957","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["syuan03/mm-ifengine"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/sft-or-rl-an-early-investigation-into","slug":"sft-or-rl-an-early-investigation-into","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","date":"2025-04-10","arxiv_id":"2504.11468","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":null}},{"paper":null,"slug":"pami-vdpo-mitigating-video-hallucinations-by","title":"PaMi-VDPO: Mitigating Video Hallucinations by Prompt-Aware Multi-Instance Video Preference Learning","date":"2025-04-08","arxiv_id":"2504.05810","n_code_links":0,"syntology":null},{"paper":null,"slug":"opencodeinstruct-a-large-scale-instruction","title":"OpenCodeInstruct: A Large-scale Instruction Tuning Dataset for Code LLMs","date":"2025-04-05","arxiv_id":"2504.04030","n_code_links":0,"syntology":null},{"paper":"/paper/anesbench-multi-dimensional-evaluation-of-llm","slug":"anesbench-multi-dimensional-evaluation-of-llm","title":"AnesBench: Multi-Dimensional Evaluation of LLM Reasoning in Anesthesiology","date":"2025-04-03","arxiv_id":"2504.02404","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mililab/anesbench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":null,"slug":"opencodereasoning-advancing-data-distillation","title":"OpenCodeReasoning: Advancing Data Distillation for Competitive Coding","date":"2025-04-02","arxiv_id":"2504.01943","n_code_links":0,"syntology":null},{"paper":"/paper/tom-rl-reinforcement-learning-unlocks-theory","slug":"tom-rl-reinforcement-learning-unlocks-theory","title":"Do Theory of Mind Benchmarks Need Explicit Human-like Reasoning in Language Models?","date":"2025-04-02","arxiv_id":"2504.01698","n_code_links":1,"syntology":{"ran":11,"of":15,"n_ran_checked":9,"n_instrument":2,"unverified":4,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["bigai-ai/ToM-RL"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/crowdvlm-r1-expanding-r1-ability-to-vision","slug":"crowdvlm-r1-expanding-r1-ability-to-vision","title":"CrowdVLM-R1: Expanding R1 Ability to Vision Language Model for Crowd Counting using Fuzzy Group Relative Policy Reward","date":"2025-03-31","arxiv_id":"2504.03724","n_code_links":1,"syntology":null},{"paper":"/paper/exploring-the-effect-of-reinforcement","slug":"exploring-the-effect-of-reinforcement","title":"Exploring the Effect of Reinforcement Learning on Video Understanding: Insights from SEED-Bench-R1","date":"2025-03-31","arxiv_id":"2503.24376","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["tencentarc/seed-bench-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"judgelrm-large-reasoning-models-as-a-judge","title":"JudgeLRM: Large Reasoning Models as a Judge","date":"2025-03-31","arxiv_id":"2504.00050","n_code_links":0,"syntology":null},{"paper":"/paper/rec-r1-bridging-generative-large-language","slug":"rec-r1-bridging-generative-large-language","title":"Rec-R1: Bridging Generative Large Language Models and User-Centric Recommendation Systems via Reinforcement Learning","date":"2025-03-31","arxiv_id":"2503.24289","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["linjc16/Rec-R1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"vgrp-bench-visual-grid-reasoning-puzzle","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","date":"2025-03-29","arxiv_id":"2503.23064","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-data-scaling-trends-and-effects-in","title":"Exploring Data Scaling Trends and Effects in Reinforcement Learning from Human Feedback","date":"2025-03-28","arxiv_id":"2503.22230","n_code_links":0,"syntology":null},{"paper":"/paper/video-r1-reinforcing-video-reasoning-in-mllms","slug":"video-r1-reinforcing-video-reasoning-in-mllms","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","date":"2025-03-27","arxiv_id":"2503.21776","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tulerfeng/video-r1"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"paper":"/paper/reason-rft-reinforcement-fine-tuning-for","slug":"reason-rft-reinforcement-fine-tuning-for","title":"Reason-RFT: Reinforcement Fine-Tuning for Visual Reasoning","date":"2025-03-26","arxiv_id":"2503.20752","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":null}},{"paper":"/paper/vpo-aligning-text-to-video-generation-models","slug":"vpo-aligning-text-to-video-generation-models","title":"VPO: Aligning Text-to-Video Generation Models with Prompt Optimization","date":"2025-03-26","arxiv_id":"2503.20491","n_code_links":1,"syntology":null},{"paper":"/paper/openvlthinker-an-early-exploration-to-complex","slug":"openvlthinker-an-early-exploration-to-complex","title":"OpenVLThinker: An Early Exploration to Complex Vision-Language Reasoning via Iterative Self-Improvement","date":"2025-03-21","arxiv_id":"2503.17352","n_code_links":1,"syntology":{"ran":9,"of":21,"n_ran_checked":9,"n_instrument":0,"unverified":12,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 12 unverified","official":{"repos":["yihedeng9/openvlthinker"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":12,"ran_from_kinds":["official"]}}},{"paper":"/paper/cls-rl-image-classification-with-rule-based","slug":"cls-rl-image-classification-with-rule-based","title":"Think or Not Think: A Study of Explicit Thinking in Rule-Based Visual Reinforcement Fine-Tuning","date":"2025-03-20","arxiv_id":"2503.16188","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["minglllli/CLS-RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"grammar-and-gameplay-aligned-rl-for-game","title":"Grammar and Gameplay-aligned RL for Game Description Generation with LLMs","date":"2025-03-20","arxiv_id":"2503.15783","n_code_links":0,"syntology":null},{"paper":null,"slug":"othink-mr1-stimulating-multimodal-generalized","title":"OThink-MR1: Stimulating multimodal generalized reasoning capabilities via dynamic reinforcement learning","date":"2025-03-20","arxiv_id":"2503.16081","n_code_links":0,"syntology":null},{"paper":"/paper/cosmos-reason1-from-physical-common-sense-to","slug":"cosmos-reason1-from-physical-common-sense-to","title":"Cosmos-Reason1: From Physical Common Sense To Embodied Reasoning","date":"2025-03-18","arxiv_id":"2503.15558","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nvidia-cosmos/cosmos-reason1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-much-do-llms-learn-from-negative-examples","slug":"how-much-do-llms-learn-from-negative-examples","title":"How much do LLMs learn from negative examples?","date":"2025-03-18","arxiv_id":"2503.14391","n_code_links":1,"syntology":null},{"paper":"/paper/light-r1-curriculum-sft-dpo-and-rl-for-long","slug":"light-r1-curriculum-sft-dpo-and-rl-for-long","title":"Light-R1: Curriculum SFT, DPO and RL for Long COT from Scratch and Beyond","date":"2025-03-13","arxiv_id":"2503.10460","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":8,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["qihoo360/light-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/aligning-to-what-limits-to-rlhf-based","slug":"aligning-to-what-limits-to-rlhf-based","title":"Aligning to What? Limits to RLHF Based Alignment","date":"2025-03-12","arxiv_id":"2503.09025","n_code_links":1,"syntology":null},{"paper":null,"slug":"dast-difficulty-aware-self-training-on-large","title":"DAST: Difficulty-Aware Self-Training on Large Language Models","date":"2025-03-12","arxiv_id":"2503.09029","n_code_links":0,"syntology":null},{"paper":"/paper/alphadrive-unleashing-the-power-of-vlms-in","slug":"alphadrive-unleashing-the-power-of-vlms-in","title":"AlphaDrive: Unleashing the Power of VLMs in Autonomous Driving via Reinforcement Learning and Reasoning","date":"2025-03-10","arxiv_id":"2503.07608","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hustvl/alphadrive"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"ksod-knowledge-supplement-for-llms-on-demand","title":"KSOD: Knowledge Supplement for LLMs On Demand","date":"2025-03-10","arxiv_id":"2503.07550","n_code_links":0,"syntology":null},{"paper":"/paper/process-supervised-llm-recommenders-via-flow","slug":"process-supervised-llm-recommenders-via-flow","title":"Process-Supervised LLM Recommenders via Flow-guided Tuning","date":"2025-03-10","arxiv_id":"2503.07377","n_code_links":1,"syntology":null},{"paper":"/paper/seedream-2-0-a-native-chinese-english","slug":"seedream-2-0-a-native-chinese-english","title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","date":"2025-03-10","arxiv_id":"2503.07703","n_code_links":1,"syntology":{"ran":11,"of":13,"n_ran_checked":8,"n_instrument":3,"unverified":2,"pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":null,"slug":"gflowvlm-enhancing-multi-step-reasoning-in","title":"GFlowVLM: Enhancing Multi-step Reasoning in Vision-Language Models with Generative Flow Networks","date":"2025-03-09","arxiv_id":"2503.06514","n_code_links":0,"syntology":null},{"paper":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a","slug":"r1-zero-s-aha-moment-in-visual-reasoning-on-a","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","date":"2025-03-07","arxiv_id":"2503.05132","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["turningpoint-ai/visualthinker-r1-zero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"improving-neutral-point-of-view-text","title":"Improving Neutral Point of View Text Generation through Parameter-Efficient Reinforcement Learning and a Small-Scale High-Quality Dataset","date":"2025-03-05","arxiv_id":"2503.03654","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-jailbreaking-of-large-models-by","title":"Efficient Jailbreaking of Large Models by Freeze Training: Lower Layers Exhibit Greater Sensitivity to Harmful Content","date":"2025-02-28","arxiv_id":"2502.20952","n_code_links":0,"syntology":null},{"paper":null,"slug":"r1-t1-fully-incentivizing-translation","title":"R1-T1: Fully Incentivizing Translation Capability in LLMs via Reasoning Learning","date":"2025-02-27","arxiv_id":"2502.19735","n_code_links":0,"syntology":null},{"paper":null,"slug":"distill-not-only-data-but-also-rewards-can","title":"Distill Not Only Data but Also Rewards: Can Smaller Language Models Surpass Larger Ones?","date":"2025-02-26","arxiv_id":"2502.19557","n_code_links":0,"syntology":null},{"paper":null,"slug":"can-large-language-models-extract-customer","title":"Can Large Language Models Extract Customer Needs as well as Professional Analysts?","date":"2025-02-25","arxiv_id":"2503.01870","n_code_links":0,"syntology":null},{"paper":"/paper/discriminative-finetuning-of-generative-large","slug":"discriminative-finetuning-of-generative-large","title":"Discriminative Finetuning of Generative Large Language Models without Reward Models and Human Preference Data","date":"2025-02-25","arxiv_id":"2502.18679","n_code_links":1,"syntology":null},{"paper":null,"slug":"value-value-aware-large-language-model-for","title":"VALUE: Value-Aware Large Language Model for Query Rewriting via Weighted Trie in Sponsored Search","date":"2025-02-25","arxiv_id":"2504.05321","n_code_links":0,"syntology":null},{"paper":"/paper/longwriter-v-enabling-ultra-long-and-high","slug":"longwriter-v-enabling-ultra-long-and-high","title":"LongWriter-V: Enabling Ultra-Long and High-Fidelity Generation in Vision-Language Models","date":"2025-02-20","arxiv_id":"2502.14834","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["thu-keg/longwriter-v"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/longpo-long-context-self-evolution-of-large","slug":"longpo-long-context-self-evolution-of-large","title":"LongPO: Long Context Self-Evolution of Large Language Models through Short-to-Long Preference Optimization","date":"2025-02-19","arxiv_id":"2502.13922","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DAMO-NLP-SG/LongPO"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improved-fine-tuning-of-large-multimodal-1","slug":"improved-fine-tuning-of-large-multimodal-1","title":"Robust Adaptation of Large Multimodal Models for Retrieval Augmented Hateful Meme Detection","date":"2025-02-18","arxiv_id":"2502.13061","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["JingbiaoMei/RGCL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"an-efficient-row-based-sparse-fine-tuning","title":"An Efficient Row-Based Sparse Fine-Tuning","date":"2025-02-17","arxiv_id":"2502.11439","n_code_links":0,"syntology":null},{"paper":"/paper/thinking-preference-optimization","slug":"thinking-preference-optimization","title":"Thinking Preference Optimization","date":"2025-02-17","arxiv_id":"2502.13173","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["uservan/ThinkPO"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"balancing-the-budget-understanding-trade-offs","title":"Balancing the Budget: Understanding Trade-offs Between Supervised and Preference-Based Finetuning","date":"2025-02-16","arxiv_id":"2502.11284","n_code_links":0,"syntology":null},{"paper":null,"slug":"simplify-rlhf-as-reward-weighted-sft-a","title":"Simplify RLHF as Reward-Weighted SFT: A Variational Method","date":"2025-02-16","arxiv_id":"2502.11026","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-diffusion-models","slug":"large-language-diffusion-models","title":"Large Language Diffusion Models","date":"2025-02-14","arxiv_id":"2502.09992","n_code_links":2,"syntology":{"ran":6,"of":7,"n_ran_checked":1,"n_instrument":5,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"preference-learning-made-easy-everything","title":"Preference learning made easy: Everything should be understood through win rate","date":"2025-02-14","arxiv_id":"2502.10505","n_code_links":0,"syntology":null},{"paper":null,"slug":"selective-self-to-supervised-fine-tuning-for","title":"Selective Self-to-Supervised Fine-Tuning for Generalization in Large Language Models","date":"2025-02-12","arxiv_id":"2502.08130","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-oversight-for-superhuman-ai-via","title":"Scalable Oversight for Superhuman AI via Recursive Self-Critiquing","date":"2025-02-07","arxiv_id":"2502.04675","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-best-instruction-tuning-data-are-those","title":"The Best Instruction-Tuning Data are Those That Fit","date":"2025-02-06","arxiv_id":"2502.04194","n_code_links":0,"syntology":null},{"paper":"/paper/demystifying-long-chain-of-thought-reasoning","slug":"demystifying-long-chain-of-thought-reasoning","title":"Demystifying Long Chain-of-Thought Reasoning in LLMs","date":"2025-02-05","arxiv_id":"2502.03373","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":5,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["eddycmu/demystify-long-cot"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/limo-less-is-more-for-reasoning","slug":"limo-less-is-more-for-reasoning","title":"LIMO: Less is More for Reasoning","date":"2025-02-05","arxiv_id":"2502.03387","n_code_links":3,"syntology":null},{"paper":null,"slug":"diffusion-instruction-tuning","title":"Diffusion Instruction Tuning","date":"2025-02-04","arxiv_id":"2502.06814","n_code_links":0,"syntology":null},{"paper":null,"slug":"token-cleaning-fine-grained-data-selection","title":"Token Cleaning: Fine-Grained Data Selection for LLM Supervised Fine-Tuning","date":"2025-02-04","arxiv_id":"2502.01968","n_code_links":0,"syntology":null},{"paper":"/paper/process-reinforcement-through-implicit","slug":"process-reinforcement-through-implicit","title":"Process Reinforcement through Implicit Rewards","date":"2025-02-03","arxiv_id":"2502.01456","n_code_links":5,"syntology":{"ran":14,"of":17,"n_ran_checked":13,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["prime-rl/prime"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"the-differences-between-direct-alignment","title":"The Differences Between Direct Alignment Algorithms are a Blur","date":"2025-02-03","arxiv_id":"2502.01237","n_code_links":0,"syntology":null},{"paper":"/paper/guardreasoner-towards-reasoning-based-llm","slug":"guardreasoner-towards-reasoning-based-llm","title":"GuardReasoner: Towards Reasoning-based LLM Safeguards","date":"2025-01-30","arxiv_id":"2501.18492","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yueliu1999/guardreasoner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/wildchat-50m-a-deep-dive-into-the-role-of","slug":"wildchat-50m-a-deep-dive-into-the-role-of","title":"WILDCHAT-50M: A Deep Dive Into the Role of Synthetic Data in Post-Training","date":"2025-01-30","arxiv_id":"2501.18511","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["penfever/wildchat-50m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/critique-fine-tuning-learning-to-critique-is","slug":"critique-fine-tuning-learning-to-critique-is","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","date":"2025-01-29","arxiv_id":"2501.17703","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"challenges-in-ensuring-ai-safety-in-deepseek","title":"Challenges in Ensuring AI Safety in DeepSeek-R1 Models: The Shortcomings of Reinforcement Learning Strategies","date":"2025-01-28","arxiv_id":"2501.17030","n_code_links":0,"syntology":null},{"paper":null,"slug":"sft-memorizes-rl-generalizes-a-comparative","title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","date":"2025-01-28","arxiv_id":"2501.17161","n_code_links":0,"syntology":null},{"paper":null,"slug":"advancing-math-reasoning-in-language-models","title":"Advancing Mathematical Reasoning in Language Models: The Impact of Problem-Solving Data, Data Synthesis Methods, and Training Stages","date":"2025-01-23","arxiv_id":"2501.14002","n_code_links":0,"syntology":null},{"paper":"/paper/videollama-3-frontier-multimodal-foundation","slug":"videollama-3-frontier-multimodal-foundation","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","date":"2025-01-22","arxiv_id":"2501.13106","n_code_links":1,"syntology":{"ran":8,"of":14,"n_ran_checked":3,"n_instrument":5,"unverified":6,"pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 6 unverified","official":{"repos":["damo-nlp-sg/videollama3"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/condor-enhance-llm-alignment-with-knowledge","slug":"condor-enhance-llm-alignment-with-knowledge","title":"Condor: Enhance LLM Alignment with Knowledge-Driven Data Synthesis and Refinement","date":"2025-01-21","arxiv_id":"2501.12273","n_code_links":1,"syntology":null},{"paper":"/paper/from-drafts-to-answers-unlocking-llm","slug":"from-drafts-to-answers-unlocking-llm","title":"From Drafts to Answers: Unlocking LLM Potential via Aggregation Fine-Tuning","date":"2025-01-21","arxiv_id":"2501.11877","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-xllms-understand-the-structure-of-dialog","title":"Can MLLMs Generalize to Multi-Party dialog? Exploring Multilingual Response Generation in Complex Scenarios","date":"2025-01-20","arxiv_id":"2501.11269","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-sar-object-detection-with-self","title":"Enhancing SAR Object Detection with Self-Supervised Pre-training on Masked Auto-Encoders","date":"2025-01-20","arxiv_id":"2501.11249","n_code_links":0,"syntology":null},{"paper":null,"slug":"boosting-tool-use-of-large-language-models","title":"iTool: Boosting Tool Use of Large Language Models via Iterative Reinforced Fine-Tuning","date":"2025-01-15","arxiv_id":"2501.09766","n_code_links":0,"syntology":null},{"paper":"/paper/iterative-label-refinement-matters-more-than","slug":"iterative-label-refinement-matters-more-than","title":"Iterative Label Refinement Matters More than Preference Optimization under Weak Supervision","date":"2025-01-14","arxiv_id":"2501.07886","n_code_links":1,"syntology":null},{"paper":null,"slug":"scalable-vision-language-model-training-via","title":"Scalable Vision Language Model Training via High Quality Data Curation","date":"2025-01-10","arxiv_id":"2501.05952","n_code_links":0,"syntology":null},{"paper":"/paper/are-they-the-same-exploring-visual","slug":"are-they-the-same-exploring-visual","title":"Are They the Same? Exploring Visual Correspondence Shortcomings of Multimodal LLMs","date":"2025-01-08","arxiv_id":"2501.04670","n_code_links":1,"syntology":null},{"paper":null,"slug":"livecc-learning-video-llm-with-streaming","title":"LiveCC: Learning Video LLM with Streaming Speech Transcription at Scale","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/natural-language-fine-tuning","slug":"natural-language-fine-tuning","title":"Natural Language Fine-Tuning","date":"2024-12-29","arxiv_id":"2412.20382","n_code_links":1,"syntology":null},{"paper":"/paper/confidence-v-s-critique-a-decomposition-of","slug":"confidence-v-s-critique-a-decomposition-of","title":"Confidence v.s. Critique: A Decomposition of Self-Correction Capability for LLMs","date":"2024-12-27","arxiv_id":"2412.19513","n_code_links":1,"syntology":{"ran":1,"of":5,"n_ran_checked":0,"n_instrument":1,"unverified":4,"pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["Zhe-Young/SelfCorrectDecompose"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-multi-step-reasoning-abilities-of","title":"Improving Multi-Step Reasoning Abilities of Large Language Models with Direct Advantage Policy Optimization","date":"2024-12-24","arxiv_id":"2412.18279","n_code_links":0,"syntology":null},{"paper":"/paper/mulberry-empowering-mllm-with-o1-like","slug":"mulberry-empowering-mllm-with-o1-like","title":"Mulberry: Empowering MLLM with o1-like Reasoning and Reflection via Collective Monte Carlo Tree Search","date":"2024-12-24","arxiv_id":"2412.18319","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["hjyao00/mulberry"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/tl-training-a-task-feature-based-framework","slug":"tl-training-a-task-feature-based-framework","title":"TL-Training: A Task-Feature-Based Framework for Training Large Language Models in Tool Use","date":"2024-12-20","arxiv_id":"2412.15495","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-to-generate-research-idea-with","title":"Learning to Generate Research Idea with Dynamic Control","date":"2024-12-19","arxiv_id":"2412.14626","n_code_links":0,"syntology":null},{"paper":null,"slug":"northeastern-uni-at-multilingual","title":"Northeastern Uni at Multilingual Counterspeech Generation: Enhancing Counter Speech Generation with LLM Alignment through Direct Preference Optimization","date":"2024-12-19","arxiv_id":"2412.15453","n_code_links":0,"syntology":null},{"paper":"/paper/pa-rag-rag-alignment-via-multi-perspective","slug":"pa-rag-rag-alignment-via-multi-perspective","title":"PA-RAG: RAG Alignment via Multi-Perspective Preference Optimization","date":"2024-12-19","arxiv_id":"2412.14510","n_code_links":1,"syntology":null},{"paper":"/paper/prompt-a-video-prompt-your-video-diffusion","slug":"prompt-a-video-prompt-your-video-diffusion","title":"Prompt-A-Video: Prompt Your Video Diffusion Model via Preference-Aligned LLM","date":"2024-12-19","arxiv_id":"2412.15156","n_code_links":1,"syntology":null},{"paper":"/paper/robustft-robust-supervised-fine-tuning-for","slug":"robustft-robust-supervised-fine-tuning-for","title":"RobustFT: Robust Supervised Fine-tuning for Large Language Models under Noisy Response","date":"2024-12-19","arxiv_id":"2412.14922","n_code_links":1,"syntology":null},{"paper":null,"slug":"seeking-consistent-flat-minima-for-better","title":"Seeking Consistent Flat Minima for Better Domain Generalization via Refining Loss Landscapes","date":"2024-12-18","arxiv_id":"2412.13573","n_code_links":0,"syntology":null},{"paper":"/paper/preference-oriented-supervised-fine-tuning","slug":"preference-oriented-supervised-fine-tuning","title":"Preference-Oriented Supervised Fine-Tuning: Favoring Target Model Over Aligned Large Language Models","date":"2024-12-17","arxiv_id":"2412.12865","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Savannah120/alignment-handbook-PoFT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/route-robust-multitask-tuning-and","slug":"route-robust-multitask-tuning-and","title":"ROUTE: Robust Multitask Tuning and Collaboration for Text-to-SQL","date":"2024-12-13","arxiv_id":"2412.10138","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alibaba/route"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"carebot-a-pioneering-full-process-open-source","title":"CareBot: A Pioneering Full-Process Open-Source Medical Language Model","date":"2024-12-12","arxiv_id":"2412.15236","n_code_links":0,"syntology":null},{"paper":"/paper/sprec-leveraging-self-play-to-debias","slug":"sprec-leveraging-self-play-to-debias","title":"SPRec: Leveraging Self-Play to Debias Preference Alignment for Large Language Model-based Recommendations","date":"2024-12-12","arxiv_id":"2412.09243","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":5,"n_instrument":2,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["regionch/sprec"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-comparative-study-of-learning-paradigms-in","title":"A Comparative Study of Learning Paradigms in Large Language Models via Intrinsic Dimension","date":"2024-12-09","arxiv_id":"2412.06245","n_code_links":0,"syntology":null},{"paper":null,"slug":"alma-alignment-with-minimal-annotation","title":"ALMA: Alignment with Minimal Annotation","date":"2024-12-05","arxiv_id":"2412.04305","n_code_links":0,"syntology":null}],"record_sha256":"3dd1d1d07a6faf93f89f9f7e7a9af29ee5f99c59ed0f3ebcd3c8a06a974283f6","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}