{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/instruction-following/papers/2","list_of":"/task/instruction-following","task":"Instruction Following","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":12,"rows_per_page":100,"rows":[101,200],"of":1135,"counts":{"archive_papers_tagged":1135,"with_a_code_link":609,"where_syntology_ran_a_sample":311,"not_listed_spam_title":0,"listed":1135,"listed_where_code_ran":311,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":255,"every_run_a_failure_of_syntologys_instrument":56,"listed_with_a_run_with_no_instrument_failure":255,"listed_every_run_a_failure_of_syntologys_instrument":56,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/instruction-following","prev":"/task/instruction-following","next":"/task/instruction-following/papers/3","papers":[{"url":"/paper/kwai-keye-vl-technical-report","slug":"kwai-keye-vl-technical-report","title":"Kwai Keye-VL Technical Report","date":"2025-07-02","arxiv_id":"2507.01949","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kwai-keye-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.01949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01949"}},"official":{"repos":["kwai-keye/keye"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-pose-enhancing-human-pose-and-action","slug":"llava-pose-enhancing-human-pose-and-action","title":"LLaVA-Pose: Enhancing Human Pose and Action Understanding via Keypoint-Integrated Instruction Tuning","date":"2025-06-26","arxiv_id":"2506.21317","repositories_listed":1,"syntology":null},{"url":"/paper/instructttseval-benchmarking-complex-natural","slug":"instructttseval-benchmarking-complex-natural","title":"InstructTTSEval: Benchmarking Complex Natural-Language Instruction Following in Text-to-Speech Systems","date":"2025-06-19","arxiv_id":"2506.16381","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-hierarchical-latent-capabilities","slug":"discovering-hierarchical-latent-capabilities","title":"Discovering Hierarchical Latent Capabilities of Language Models via Causal Representation Learning","date":"2025-06-12","arxiv_id":"2506.10378","repositories_listed":1,"syntology":null},{"url":"/paper/verif-verification-engineering-for","slug":"verif-verification-engineering-for","title":"VerIF: Verification Engineering for Reinforcement Learning in Instruction Following","date":"2025-06-11","arxiv_id":"2506.09942","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/verif-verification-engineering-for#ran","syntology_url":"https://syntology.ai/paper/2506.09942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09942"}},"official":{"repos":["thu-keg/verif"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eifbench-extremely-complex-instruction","slug":"eifbench-extremely-complex-instruction","title":"EIFBENCH: Extremely Complex Instruction Following Benchmark for Large Language Models","date":"2025-06-10","arxiv_id":"2506.08375","repositories_listed":1,"syntology":null},{"url":"/paper/levo-high-quality-song-generation-with-multi","slug":"levo-high-quality-song-generation-with-multi","title":"LeVo: High-Quality Song Generation with Multi-Preference Alignment","date":"2025-06-09","arxiv_id":"2506.07520","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/levo-high-quality-song-generation-with-multi#ran","syntology_url":"https://syntology.ai/paper/2506.07520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07520"}},"official":{"repos":["tencent-ailab/songgeneration"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/adversarial-paraphrasing-a-universal-attack","slug":"adversarial-paraphrasing-a-universal-attack","title":"Adversarial Paraphrasing: A Universal Attack for Humanizing AI-Generated Text","date":"2025-06-08","arxiv_id":"2506.07001","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 5 unverified","sample_list":"/paper/adversarial-paraphrasing-a-universal-attack#ran","syntology_url":"https://syntology.ai/paper/2506.07001","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07001"}},"official":{"repos":["chengez/adversarial-paraphrasing"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/being-strong-progressively-enhancing","slug":"being-strong-progressively-enhancing","title":"Being Strong Progressively! Enhancing Knowledge Distillation of Large Language Models through a Curriculum Learning Framework","date":"2025-06-06","arxiv_id":"2506.05695","repositories_listed":1,"syntology":null},{"url":"/paper/identifying-reliable-evaluation-metrics-for","slug":"identifying-reliable-evaluation-metrics-for","title":"Identifying Reliable Evaluation Metrics for Scientific Text Revision","date":"2025-06-05","arxiv_id":"2506.04772","repositories_listed":1,"syntology":null},{"url":"/paper/rewardanything-generalizable-principle","slug":"rewardanything-generalizable-principle","title":"RewardAnything: Generalizable Principle-Following Reward Models","date":"2025-06-04","arxiv_id":"2506.03637","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-reasoning-for-advanced","slug":"incentivizing-reasoning-for-advanced","title":"Incentivizing Reasoning for Advanced Instruction-Following of Large Language Models","date":"2025-06-02","arxiv_id":"2506.01413","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/incentivizing-reasoning-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2506.01413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01413"}},"official":{"repos":["yuleiqin/raif"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rewardbench-2-advancing-reward-model","slug":"rewardbench-2-advancing-reward-model","title":"RewardBench 2: Advancing Reward Model Evaluation","date":"2025-06-02","arxiv_id":"2506.01937","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rewardbench-2-advancing-reward-model#ran","syntology_url":"https://syntology.ai/paper/2506.01937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01937"}},"official":null}},{"url":"/paper/fusionaudio-1-2m-towards-fine-grained-audio","slug":"fusionaudio-1-2m-towards-fine-grained-audio","title":"FusionAudio-1.2M: Towards Fine-grained Audio Captioning with Multimodal Contextual Fusion","date":"2025-06-01","arxiv_id":"2506.01111","repositories_listed":1,"syntology":null},{"url":"/paper/don-t-reinvent-the-wheel-efficient","slug":"don-t-reinvent-the-wheel-efficient","title":"Don't Reinvent the Wheel: Efficient Instruction-Following Text Embedding based on Guided Space Transformation","date":"2025-05-30","arxiv_id":"2505.24754","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/don-t-reinvent-the-wheel-efficient#ran","syntology_url":"https://syntology.ai/paper/2505.24754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24754"}},"official":{"repos":["yingchaojiefeng/gstransform"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-large-multimodal-models-confront","slug":"when-large-multimodal-models-confront","title":"When Large Multimodal Models Confront Evolving Knowledge:Challenges and Pathways","date":"2025-05-30","arxiv_id":"2505.24449","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/when-large-multimodal-models-confront#ran","syntology_url":"https://syntology.ai/paper/2505.24449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24449"}},"official":{"repos":["EVOKE-LMM/EVOKE"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/let-them-talk-audio-driven-multi-person","slug":"let-them-talk-audio-driven-multi-person","title":"Let Them Talk: Audio-Driven Multi-Person Conversational Video Generation","date":"2025-05-28","arxiv_id":"2505.22647","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-them-talk-audio-driven-multi-person#ran","syntology_url":"https://syntology.ai/paper/2505.22647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22647"}},"official":{"repos":["meigen-ai/multitalk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-course-correction-in-steerability","slug":"a-course-correction-in-steerability","title":"A Course Correction in Steerability Evaluation: Revealing Miscalibration and Side Effects in LLMs","date":"2025-05-27","arxiv_id":"2505.23816","repositories_listed":1,"syntology":null},{"url":"/paper/speech-ifeval-evaluating-instruction","slug":"speech-ifeval-evaluating-instruction","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","date":"2025-05-25","arxiv_id":"2505.19037","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speech-ifeval-evaluating-instruction#ran","syntology_url":"https://syntology.ai/paper/2505.19037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19037"}},"official":{"repos":["kehanlu/speech-ifeval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/strict-stress-test-of-rendering-images","slug":"strict-stress-test-of-rendering-images","title":"STRICT: Stress Test of Rendering Images Containing Text","date":"2025-05-25","arxiv_id":"2505.18985","repositories_listed":1,"syntology":null},{"url":"/paper/omnigenbench-a-benchmark-for-omnipotent","slug":"omnigenbench-a-benchmark-for-omnipotent","title":"OmniGenBench: A Benchmark for Omnipotent Multimodal Generation across 50+ Tasks","date":"2025-05-24","arxiv_id":"2505.18775","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-transport-based-token-weighting","slug":"optimal-transport-based-token-weighting","title":"Optimal Transport-Based Token Weighting scheme for Enhanced Preference Optimization","date":"2025-05-24","arxiv_id":"2505.18720","repositories_listed":1,"syntology":null},{"url":"/paper/ida-bench-evaluating-llms-on-interactive","slug":"ida-bench-evaluating-llms-on-interactive","title":"IDA-Bench: Evaluating LLMs on Interactive Guided Data Analysis","date":"2025-05-23","arxiv_id":"2505.18223","repositories_listed":1,"syntology":null},{"url":"/paper/agentif-benchmarking-instruction-following-of","slug":"agentif-benchmarking-instruction-following-of","title":"AGENTIF: Benchmarking Instruction Following of Large Language Models in Agentic Scenarios","date":"2025-05-22","arxiv_id":"2505.16944","repositories_listed":1,"syntology":null},{"url":"/paper/castillo-characterizing-response-length","slug":"castillo-characterizing-response-length","title":"CASTILLO: Characterizing Response Length Distributions of Large Language Models","date":"2025-05-22","arxiv_id":"2505.16881","repositories_listed":1,"syntology":null},{"url":"/paper/ifeval-audio-benchmarking-instruction","slug":"ifeval-audio-benchmarking-instruction","title":"IFEval-Audio: Benchmarking Instruction-Following Capability in Audio-based Large Language Models","date":"2025-05-22","arxiv_id":"2505.16774","repositories_listed":1,"syntology":null},{"url":"/paper/lavida-a-large-diffusion-language-model-for","slug":"lavida-a-large-diffusion-language-model-for","title":"LaViDa: A Large Diffusion Language Model for Multimodal Understanding","date":"2025-05-22","arxiv_id":"2505.16839","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lavida-a-large-diffusion-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2505.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16839"}},"official":{"repos":["jacklishufan/lavida"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/lifebench-evaluating-length-instruction","slug":"lifebench-evaluating-length-instruction","title":"LIFEBench: Evaluating Length Instruction Following in Large Language Models","date":"2025-05-22","arxiv_id":"2505.16234","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-reasoning-losing-control-evaluating","slug":"scaling-reasoning-losing-control-evaluating","title":"Scaling Reasoning, Losing Control: Evaluating Instruction Following in Large Reasoning Models","date":"2025-05-20","arxiv_id":"2505.14810","repositories_listed":1,"syntology":null},{"url":"/paper/what-prompts-don-t-say-understanding-and","slug":"what-prompts-don-t-say-understanding-and","title":"What Prompts Don't Say: Understanding and Managing Underspecification in LLM Prompts","date":"2025-05-19","arxiv_id":"2505.13360","repositories_listed":1,"syntology":null},{"url":"/paper/internal-causal-mechanisms-robustly-predict","slug":"internal-causal-mechanisms-robustly-predict","title":"Internal Causal Mechanisms Robustly Predict Language Model Out-of-Distribution Behaviors","date":"2025-05-17","arxiv_id":"2505.11770","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internal-causal-mechanisms-robustly-predict#ran","syntology_url":"https://syntology.ai/paper/2505.11770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11770"}},"official":{"repos":["explanare/ood-prediction"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10833","slug":"2505-10833","title":"MergeBench: A Benchmark for Merging Domain-Specialized LLMs","date":"2025-05-16","arxiv_id":"2505.10833","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11080","slug":"2505-11080","title":"BLEUBERI: BLEU is a surprisingly effective reward for instruction following","date":"2025-05-16","arxiv_id":"2505.11080","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11493","slug":"2505-11493","title":"GIE-Bench: Towards Grounded Evaluation for Text-Guided Image Editing","date":"2025-05-16","arxiv_id":"2505.11493","repositories_listed":1,"syntology":null},{"url":"/paper/healthbench-evaluating-large-language-models","slug":"healthbench-evaluating-large-language-models","title":"HealthBench: Evaluating Large Language Models Towards Improved Human Health","date":"2025-05-13","arxiv_id":"2505.08775","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/healthbench-evaluating-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2505.08775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.08775"}},"official":{"repos":["openai/simple-evals"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/judging-the-judges-can-large-vision-language","slug":"judging-the-judges-can-large-vision-language","title":"Judging the Judges: Can Large Vision-Language Models Fairly Evaluate Chart Comprehension and Reasoning?","date":"2025-05-13","arxiv_id":"2505.08468","repositories_listed":1,"syntology":null},{"url":"/paper/a-multi-dimensional-constraint-framework-for","slug":"a-multi-dimensional-constraint-framework-for","title":"A Multi-Dimensional Constraint Framework for Evaluating and Improving Instruction Following in Large Language Models","date":"2025-05-12","arxiv_id":"2505.07591","repositories_listed":1,"syntology":null},{"url":"/paper/mm-skin-enhancing-dermatology-vision-language","slug":"mm-skin-enhancing-dermatology-vision-language","title":"MM-Skin: Enhancing Dermatology Vision-Language Model with an Image-Text Dataset Derived from Textbooks","date":"2025-05-09","arxiv_id":"2505.06152","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-markup-language-generation-for","slug":"adaptive-markup-language-generation-for","title":"Adaptive Markup Language Generation for Contextually-Grounded Visual Document Understanding","date":"2025-05-08","arxiv_id":"2505.05446","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/adaptive-markup-language-generation-for#ran","syntology_url":"https://syntology.ai/paper/2505.05446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05446"}},"official":{"repos":["Euphoria16/DocMark"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/tf1-en-3m-three-million-synthetic-moral","slug":"tf1-en-3m-three-million-synthetic-moral","title":"TF1-EN-3M: Three Million Synthetic Moral Fables for Training Small, Open Language Models","date":"2025-04-29","arxiv_id":"2504.20605","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-tuning-data-synthesis-from","slug":"instruction-tuning-data-synthesis-from","title":"Instruction-Tuning Data Synthesis from Scratch via Web Reconstruction","date":"2025-04-22","arxiv_id":"2504.15573","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chinese-vicuna-a-chinese-instruction","slug":"chinese-vicuna-a-chinese-instruction","title":"Chinese-Vicuna: A Chinese Instruction-following Llama-based Model","date":"2025-04-17","arxiv_id":"2504.12737","repositories_listed":1,"syntology":null},{"url":"/paper/a-dual-space-framework-for-general-knowledge","slug":"a-dual-space-framework-for-general-knowledge","title":"A Dual-Space Framework for General Knowledge Distillation of Large Language Models","date":"2025-04-15","arxiv_id":"2504.11426","repositories_listed":1,"syntology":null},{"url":"/paper/realwebassist-a-benchmark-for-long-horizon","slug":"realwebassist-a-benchmark-for-long-horizon","title":"RealWebAssist: A Benchmark for Long-Horizon Web Assistance with Real-World Users","date":"2025-04-14","arxiv_id":"2504.10445","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/realwebassist-a-benchmark-for-long-horizon#ran","syntology_url":"https://syntology.ai/paper/2504.10445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10445"}},"official":{"repos":["scai-jhu/realwebassist"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/playpen-an-environment-for-exploring-learning","slug":"playpen-an-environment-for-exploring-learning","title":"Playpen: An Environment for Exploring Learning Through Conversational Interaction","date":"2025-04-11","arxiv_id":"2504.08590","repositories_listed":1,"syntology":null},{"url":"/paper/mm-ifengine-towards-multimodal-instruction","slug":"mm-ifengine-towards-multimodal-instruction","title":"MM-IFEngine: Towards Multimodal Instruction Following","date":"2025-04-10","arxiv_id":"2504.07957","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mm-ifengine-towards-multimodal-instruction#ran","syntology_url":"https://syntology.ai/paper/2504.07957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07957"}},"official":{"repos":["syuan03/mm-ifengine"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sculpting-subspaces-constrained-full-fine","slug":"sculpting-subspaces-constrained-full-fine","title":"Sculpting Subspaces: Constrained Full Fine-Tuning in LLMs for Continual Learning","date":"2025-04-09","arxiv_id":"2504.07097","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-single-turn-a-survey-on-multi-turn","slug":"beyond-single-turn-a-survey-on-multi-turn","title":"Beyond Single-Turn: A Survey on Multi-Turn Interactions with Large Language Models","date":"2025-04-07","arxiv_id":"2504.04717","repositories_listed":1,"syntology":null},{"url":"/paper/crystalformer-rl-reinforcement-fine-tuning","slug":"crystalformer-rl-reinforcement-fine-tuning","title":"CrystalFormer-RL: Reinforcement Fine-Tuning for Materials Design","date":"2025-04-03","arxiv_id":"2504.02367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crystalformer-rl-reinforcement-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2504.02367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02367"}},"official":{"repos":["deepmodeling/crystalformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sting-bee-towards-vision-language-model-for","slug":"sting-bee-towards-vision-language-model-for","title":"STING-BEE: Towards Vision-Language Model for Real-World X-ray Baggage Security Inspection","date":"2025-04-03","arxiv_id":"2504.02823","repositories_listed":1,"syntology":null},{"url":"/paper/rec-r1-bridging-generative-large-language","slug":"rec-r1-bridging-generative-large-language","title":"Rec-R1: Bridging Generative Large Language Models and User-Centric Recommendation Systems via Reinforcement Learning","date":"2025-03-31","arxiv_id":"2503.24289","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rec-r1-bridging-generative-large-language#ran","syntology_url":"https://syntology.ai/paper/2503.24289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24289"}},"official":{"repos":["linjc16/Rec-R1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/insvie-1m-effective-instruction-based-video","slug":"insvie-1m-effective-instruction-based-video","title":"InsViE-1M: Effective Instruction-based Video Editing with Elaborate Dataset Construction","date":"2025-03-26","arxiv_id":"2503.20287","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/insvie-1m-effective-instruction-based-video#ran","syntology_url":"https://syntology.ai/paper/2503.20287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20287"}},"official":{"repos":["langmanbusi/insvie"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/qwen2-5-omni-technical-report","slug":"qwen2-5-omni-technical-report","title":"Qwen2.5-Omni Technical Report","date":"2025-03-26","arxiv_id":"2503.20215","repositories_listed":1,"syntology":null},{"url":"/paper/simplerl-zoo-investigating-and-taming-zero","slug":"simplerl-zoo-investigating-and-taming-zero","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","date":"2025-03-24","arxiv_id":"2503.18892","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplerl-zoo-investigating-and-taming-zero#ran","syntology_url":"https://syntology.ai/paper/2503.18892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18892"}},"official":null}},{"url":"/paper/does-context-matter-contextualjudgebench-for","slug":"does-context-matter-contextualjudgebench-for","title":"Does Context Matter? ContextualJudgeBench for Evaluating LLM-based Judges in Contextual Settings","date":"2025-03-19","arxiv_id":"2503.15620","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-context-matter-contextualjudgebench-for#ran","syntology_url":"https://syntology.ai/paper/2503.15620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15620"}},"official":{"repos":["salesforceairesearch/contextualjudgebench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-more-a-comparative-study-of-llms-and","slug":"llava-more-a-comparative-study-of-llms-and","title":"LLaVA-MORE: A Comparative Study of LLMs and Visual Backbones for Enhanced Visual Instruction Tuning","date":"2025-03-19","arxiv_id":"2503.15621","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/llava-more-a-comparative-study-of-llms-and#ran","syntology_url":"https://syntology.ai/paper/2503.15621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15621"}},"official":{"repos":["aimagelab/LLaVA-MORE"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/can-language-models-follow-multiple-turns-of","slug":"can-language-models-follow-multiple-turns-of","title":"Can Language Models Follow Multiple Turns of Entangled Instructions?","date":"2025-03-17","arxiv_id":"2503.13222","repositories_listed":1,"syntology":null},{"url":"/paper/asma-tune-unlocking-llms-assembly-code","slug":"asma-tune-unlocking-llms-assembly-code","title":"ASMA-Tune: Unlocking LLMs' Assembly Code Comprehension via Structural-Semantic Instruction Tuning","date":"2025-03-14","arxiv_id":"2503.11617","repositories_listed":1,"syntology":null},{"url":"/paper/robust-multi-objective-controlled-decoding-of","slug":"robust-multi-objective-controlled-decoding-of","title":"Robust Multi-Objective Controlled Decoding of Large Language Models","date":"2025-03-11","arxiv_id":"2503.08796","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-multi-objective-controlled-decoding-of#ran","syntology_url":"https://syntology.ai/paper/2503.08796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08796"}},"official":{"repos":["williambankes/robust-multi-objective-decoding"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distillm-2-a-contrastive-approach-boosts-the","slug":"distillm-2-a-contrastive-approach-boosts-the","title":"DistiLLM-2: A Contrastive Approach Boosts the Distillation of LLMs","date":"2025-03-10","arxiv_id":"2503.07067","repositories_listed":1,"syntology":null},{"url":"/paper/ref-vlm-triplet-based-referring-paradigm-for","slug":"ref-vlm-triplet-based-referring-paradigm-for","title":"REF-VLM: Triplet-Based Referring Paradigm for Unified Visual Decoding","date":"2025-03-10","arxiv_id":"2503.07413","repositories_listed":1,"syntology":null},{"url":"/paper/seedream-2-0-a-native-chinese-english","slug":"seedream-2-0-a-native-chinese-english","title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","date":"2025-03-10","arxiv_id":"2503.07703","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/seedream-2-0-a-native-chinese-english#ran","syntology_url":"https://syntology.ai/paper/2503.07703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07703"}},"official":null}},{"url":"/paper/wildifeval-instruction-following-in-the-wild","slug":"wildifeval-instruction-following-in-the-wild","title":"WildIFEval: Instruction Following in the Wild","date":"2025-03-09","arxiv_id":"2503.06573","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wildifeval-instruction-following-in-the-wild#ran","syntology_url":"https://syntology.ai/paper/2503.06573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06573"}},"official":{"repos":["gililior/wild-if-eval-code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/routereval-a-comprehensive-benchmark-for","slug":"routereval-a-comprehensive-benchmark-for","title":"RouterEval: A Comprehensive Benchmark for Routing LLMs to Explore Model-level Scaling Up in LLMs","date":"2025-03-08","arxiv_id":"2503.10657","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/routereval-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2503.10657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10657"}},"official":{"repos":["milkthink-lab/routereval"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fusechat-3-0-preference-optimization-meets","slug":"fusechat-3-0-preference-optimization-meets","title":"FuseChat-3.0: Preference Optimization Meets Heterogeneous Model Fusion","date":"2025-03-06","arxiv_id":"2503.04222","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-cross-lingual-rewarding-for","slug":"implicit-cross-lingual-rewarding-for","title":"Implicit Cross-Lingual Rewarding for Efficient Multilingual Preference Alignment","date":"2025-03-06","arxiv_id":"2503.04647","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-reasoning-queries-a-systematic","slug":"attentive-reasoning-queries-a-systematic","title":"Attentive Reasoning Queries: A Systematic Method for Optimizing Instruction-Following in Large Language Models","date":"2025-03-05","arxiv_id":"2503.03669","repositories_listed":1,"syntology":null},{"url":"/paper/crowdselect-synthetic-instruction-data","slug":"crowdselect-synthetic-instruction-data","title":"CrowdSelect: Synthetic Instruction Data Selection with Multi-LLM Wisdom","date":"2025-03-03","arxiv_id":"2503.01836","repositories_listed":1,"syntology":null},{"url":"/paper/re-imagining-multimodal-instruction-tuning-a","slug":"re-imagining-multimodal-instruction-tuning-a","title":"Re-Imagining Multimodal Instruction Tuning: A Representation View","date":"2025-03-02","arxiv_id":"2503.00723","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/re-imagining-multimodal-instruction-tuning-a#ran","syntology_url":"https://syntology.ai/paper/2503.00723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00723"}},"official":{"repos":["comeandcode/MRT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/agentic-reward-modeling-integrating-human","slug":"agentic-reward-modeling-integrating-human","title":"Agentic Reward Modeling: Integrating Human Preferences with Verifiable Correctness Signals for Reliable Reward Systems","date":"2025-02-26","arxiv_id":"2502.19328","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":3,"n_ran_checked":8,"n_instrument":4,"n_unverified":2,"n_honours":4,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"12 ran (of which 3 constructed an object rather than computing a result; 8 with no instrument failure: 4 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-reward-modeling-integrating-human#ran","syntology_url":"https://syntology.ai/paper/2502.19328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19328"}},"official":{"repos":["thu-keg/agentic-reward-modeling"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":3,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/codeif-benchmarking-the-instruction-following","slug":"codeif-benchmarking-the-instruction-following","title":"CodeIF: Benchmarking the Instruction-Following Capabilities of Large Language Models for Code Generation","date":"2025-02-26","arxiv_id":"2502.19166","repositories_listed":1,"syntology":null},{"url":"/paper/stay-focused-problem-drift-in-multi-agent","slug":"stay-focused-problem-drift-in-multi-agent","title":"Stay Focused: Problem Drift in Multi-Agent Debate","date":"2025-02-26","arxiv_id":"2502.19559","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stay-focused-problem-drift-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2502.19559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19559"}},"official":{"repos":["jonas-becker/problem-drift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rank1-test-time-compute-for-reranking-in","slug":"rank1-test-time-compute-for-reranking-in","title":"Rank1: Test-Time Compute for Reranking in Information Retrieval","date":"2025-02-25","arxiv_id":"2502.18418","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rank1-test-time-compute-for-reranking-in#ran","syntology_url":"https://syntology.ai/paper/2502.18418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18418"}},"official":{"repos":["orionw/rank1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/capability-instruction-tuning-a-new-paradigm","slug":"capability-instruction-tuning-a-new-paradigm","title":"Capability Instruction Tuning: A New Paradigm for Dynamic LLM Routing","date":"2025-02-24","arxiv_id":"2502.17282","repositories_listed":1,"syntology":null},{"url":"/paper/order-matters-investigate-the-position-bias","slug":"order-matters-investigate-the-position-bias","title":"Order Matters: Investigate the Position Bias in Multi-constraint Instruction Following","date":"2025-02-24","arxiv_id":"2502.17204","repositories_listed":1,"syntology":null},{"url":"/paper/natsgld-a-dataset-with-speech-gesture-logic","slug":"natsgld-a-dataset-with-speech-gesture-logic","title":"NatSGLD: A Dataset with Speech, Gesture, Logic, and Demonstration for Robot Learning in Natural Human-Robot Interaction","date":"2025-02-23","arxiv_id":"2502.16718","repositories_listed":1,"syntology":null},{"url":"/paper/sotopia-o-dynamic-strategy-injection-learning","slug":"sotopia-o-dynamic-strategy-injection-learning","title":"SOTOPIA-$Ω$: Dynamic Strategy Injection Learning and Social Instruction Following Evaluation for Social Agents","date":"2025-02-21","arxiv_id":"2502.15538","repositories_listed":1,"syntology":{"n":19,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":14,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":19,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/sotopia-o-dynamic-strategy-injection-learning#ran","syntology_url":"https://syntology.ai/paper/2502.15538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15538"}},"official":{"repos":["WYRipple/SOTOPIA-Omega"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/structflowbench-a-structured-flow-benchmark","slug":"structflowbench-a-structured-flow-benchmark","title":"StructFlowBench: A Structured Flow Benchmark for Multi-turn Instruction Following","date":"2025-02-20","arxiv_id":"2502.14494","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/structflowbench-a-structured-flow-benchmark#ran","syntology_url":"https://syntology.ai/paper/2502.14494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14494"}},"official":{"repos":["mlgroupjlu/structflowbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmteb-massive-multilingual-text-embedding","slug":"mmteb-massive-multilingual-text-embedding","title":"MMTEB: Massive Multilingual Text Embedding Benchmark","date":"2025-02-19","arxiv_id":"2502.13595","repositories_listed":1,"syntology":null},{"url":"/paper/tess-2-a-large-scale-generalist-diffusion","slug":"tess-2-a-large-scale-generalist-diffusion","title":"TESS 2: A Large-Scale Generalist Diffusion Language Model","date":"2025-02-19","arxiv_id":"2502.13917","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tess-2-a-large-scale-generalist-diffusion#ran","syntology_url":"https://syntology.ai/paper/2502.13917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13917"}},"official":{"repos":["hamishivi/tess-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/musc-improving-complex-instruction-following","slug":"musc-improving-complex-instruction-following","title":"MuSC: Improving Complex Instruction Following with Multi-granularity Self-Contrastive Training","date":"2025-02-17","arxiv_id":"2502.11541","repositories_listed":1,"syntology":null},{"url":"/paper/rolemrc-a-fine-grained-composite-benchmark","slug":"rolemrc-a-fine-grained-composite-benchmark","title":"RoleMRC: A Fine-Grained Composite Benchmark for Role-Playing and Instruction-Following","date":"2025-02-17","arxiv_id":"2502.11387","repositories_listed":1,"syntology":null},{"url":"/paper/step-audio-unified-understanding-and","slug":"step-audio-unified-understanding-and","title":"Step-Audio: Unified Understanding and Generation in Intelligent Speech Interaction","date":"2025-02-17","arxiv_id":"2502.11946","repositories_listed":1,"syntology":null},{"url":"/paper/cordial-can-multimodal-large-language-models","slug":"cordial-can-multimodal-large-language-models","title":"CORDIAL: Can Multimodal Large Language Models Effectively Understand Coherence Relationships?","date":"2025-02-16","arxiv_id":"2502.11300","repositories_listed":1,"syntology":null},{"url":"/paper/cuckoo-an-ie-free-rider-hatched-by-massive","slug":"cuckoo-an-ie-free-rider-hatched-by-massive","title":"Cuckoo: An IE Free Rider Hatched by Massive Nutrition in LLM's Nest","date":"2025-02-16","arxiv_id":"2502.11275","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-cross-tokenizer-knowledge","slug":"enhancing-cross-tokenizer-knowledge","title":"Enhancing Cross-Tokenizer Knowledge Distillation with Contextual Dynamical Mapping","date":"2025-02-16","arxiv_id":"2502.11104","repositories_listed":1,"syntology":null},{"url":"/paper/rewrite-to-jailbreak-discover-learnable-and","slug":"rewrite-to-jailbreak-discover-learnable-and","title":"Rewrite to Jailbreak: Discover Learnable and Transferable Implicit Harmfulness Instruction","date":"2025-02-16","arxiv_id":"2502.11084","repositories_listed":1,"syntology":null},{"url":"/paper/iheval-evaluating-language-models-on","slug":"iheval-evaluating-language-models-on","title":"IHEval: Evaluating Language Models on Following the Instruction Hierarchy","date":"2025-02-12","arxiv_id":"2502.08745","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iheval-evaluating-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2502.08745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.08745"}},"official":{"repos":["ytyz1307zzh/IHEval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmax-a-comprehensive-multilingual","slug":"benchmax-a-comprehensive-multilingual","title":"BenchMAX: A Comprehensive Multilingual Evaluation Suite for Large Language Models","date":"2025-02-11","arxiv_id":"2502.07346","repositories_listed":1,"syntology":null},{"url":"/paper/m-ifeval-multilingual-instruction-following","slug":"m-ifeval-multilingual-instruction-following","title":"M-IFEval: Multilingual Instruction-Following Evaluation","date":"2025-02-07","arxiv_id":"2502.04688","repositories_listed":1,"syntology":null},{"url":"/paper/ultraif-advancing-instruction-following-from","slug":"ultraif-advancing-instruction-following-from","title":"UltraIF: Advancing Instruction Following from the Wild","date":"2025-02-06","arxiv_id":"2502.04153","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ultraif-advancing-instruction-following-from#ran","syntology_url":"https://syntology.ai/paper/2502.04153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04153"}},"official":{"repos":["kkk-an/ultraif"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/code-blockwise-control-for-denoising","slug":"code-blockwise-control-for-denoising","title":"CoDe: Blockwise Control for Denoising Diffusion Models","date":"2025-02-03","arxiv_id":"2502.00968","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-blockwise-control-for-denoising#ran","syntology_url":"https://syntology.ai/paper/2502.00968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.00968"}},"official":{"repos":["anujinho/code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mfollowir-a-multilingual-benchmark-for","slug":"mfollowir-a-multilingual-benchmark-for","title":"mFollowIR: a Multilingual Benchmark for Instruction Following in Retrieval","date":"2025-01-31","arxiv_id":"2501.19264","repositories_listed":1,"syntology":null},{"url":"/paper/critique-fine-tuning-learning-to-critique-is","slug":"critique-fine-tuning-learning-to-critique-is","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","date":"2025-01-29","arxiv_id":"2501.17703","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critique-fine-tuning-learning-to-critique-is#ran","syntology_url":"https://syntology.ai/paper/2501.17703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17703"}},"official":null}},{"url":"/paper/janus-pro-unified-multimodal-understanding","slug":"janus-pro-unified-multimodal-understanding","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","date":"2025-01-29","arxiv_id":"2501.17811","repositories_listed":1,"syntology":null},{"url":"/paper/multichallenge-a-realistic-multi-turn","slug":"multichallenge-a-realistic-multi-turn","title":"MultiChallenge: A Realistic Multi-Turn Conversation Evaluation Benchmark Challenging to Frontier LLMs","date":"2025-01-29","arxiv_id":"2501.17399","repositories_listed":1,"syntology":null},{"url":"/paper/the-breeze-2-herd-of-models-traditional","slug":"the-breeze-2-herd-of-models-traditional","title":"The Breeze 2 Herd of Models: Traditional Chinese LLMs Based on Llama with Vision-Aware and Function-Calling Capabilities","date":"2025-01-23","arxiv_id":"2501.13921","repositories_listed":1,"syntology":null},{"url":"/paper/online-preference-alignment-for-language","slug":"online-preference-alignment-for-language","title":"Online Preference Alignment for Language Models via Count-based Exploration","date":"2025-01-22","arxiv_id":"2501.12735","repositories_listed":1,"syntology":null},{"url":"/paper/test-time-preference-optimization-on-the-fly","slug":"test-time-preference-optimization-on-the-fly","title":"Test-Time Preference Optimization: On-the-Fly Alignment via Iterative Textual Feedback","date":"2025-01-22","arxiv_id":"2501.12895","repositories_listed":1,"syntology":null}],"record_sha256":"f5cc500965c6596dbff1958db63ae7b652177f703c1c30975c76a31fb3a82d47","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}