{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/instruction-following/papers/ran/1","list_of":"/task/instruction-following","task":"Instruction Following","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":4,"rows_per_page":100,"rows":[1,100],"of":311,"counts":{"archive_papers_tagged":1135,"with_a_code_link":609,"where_syntology_ran_a_sample":311,"not_listed_spam_title":0,"listed":1135,"listed_where_code_ran":311,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":255,"every_run_a_failure_of_syntologys_instrument":56,"listed_with_a_run_with_no_instrument_failure":255,"listed_every_run_a_failure_of_syntologys_instrument":56,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/instruction-following/papers/ran/1","prev":null,"next":"/task/instruction-following/papers/ran/2","papers":[{"url":"/paper/drafterbench-benchmarking-large-language","slug":"drafterbench-benchmarking-large-language","title":"DrafterBench: Benchmarking Large Language Models for Tasks Automation in Civil Engineering","date":"2025-07-15","arxiv_id":"2507.11527","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/drafterbench-benchmarking-large-language#ran","syntology_url":"https://syntology.ai/paper/2507.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11527"}},"official":{"repos":["eason-li-ais/drafterbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-secalign-a-secure-foundation-llm-against","slug":"meta-secalign-a-secure-foundation-llm-against","title":"Meta SecAlign: A Secure Foundation LLM Against Prompt Injection Attacks","date":"2025-07-03","arxiv_id":"2507.02735","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-secalign-a-secure-foundation-llm-against#ran","syntology_url":"https://syntology.ai/paper/2507.02735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02735"}},"official":{"repos":["facebookresearch/meta_secalign"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/desta2-5-audio-toward-general-purpose-large","slug":"desta2-5-audio-toward-general-purpose-large","title":"DeSTA2.5-Audio: Toward General-Purpose Large Audio Language Model with Self-Generated Cross-Modal Alignment","date":"2025-07-03","arxiv_id":"2507.02768","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/desta2-5-audio-toward-general-purpose-large#ran","syntology_url":"https://syntology.ai/paper/2507.02768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.02768"}},"official":{"repos":["kehanlu/desta2.5-audio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kwai-keye-vl-technical-report","slug":"kwai-keye-vl-technical-report","title":"Kwai Keye-VL Technical Report","date":"2025-07-02","arxiv_id":"2507.01949","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kwai-keye-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.01949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01949"}},"official":{"repos":["kwai-keye/keye"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/verif-verification-engineering-for","slug":"verif-verification-engineering-for","title":"VerIF: Verification Engineering for Reinforcement Learning in Instruction Following","date":"2025-06-11","arxiv_id":"2506.09942","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/verif-verification-engineering-for#ran","syntology_url":"https://syntology.ai/paper/2506.09942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09942"}},"official":{"repos":["thu-keg/verif"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/levo-high-quality-song-generation-with-multi","slug":"levo-high-quality-song-generation-with-multi","title":"LeVo: High-Quality Song Generation with Multi-Preference Alignment","date":"2025-06-09","arxiv_id":"2506.07520","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/levo-high-quality-song-generation-with-multi#ran","syntology_url":"https://syntology.ai/paper/2506.07520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07520"}},"official":{"repos":["tencent-ailab/songgeneration"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/incentivizing-reasoning-for-advanced","slug":"incentivizing-reasoning-for-advanced","title":"Incentivizing Reasoning for Advanced Instruction-Following of Large Language Models","date":"2025-06-02","arxiv_id":"2506.01413","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/incentivizing-reasoning-for-advanced#ran","syntology_url":"https://syntology.ai/paper/2506.01413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01413"}},"official":{"repos":["yuleiqin/raif"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rewardbench-2-advancing-reward-model","slug":"rewardbench-2-advancing-reward-model","title":"RewardBench 2: Advancing Reward Model Evaluation","date":"2025-06-02","arxiv_id":"2506.01937","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rewardbench-2-advancing-reward-model#ran","syntology_url":"https://syntology.ai/paper/2506.01937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01937"}},"official":null}},{"url":"/paper/when-large-multimodal-models-confront","slug":"when-large-multimodal-models-confront","title":"When Large Multimodal Models Confront Evolving Knowledge:Challenges and Pathways","date":"2025-05-30","arxiv_id":"2505.24449","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/when-large-multimodal-models-confront#ran","syntology_url":"https://syntology.ai/paper/2505.24449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24449"}},"official":{"repos":["EVOKE-LMM/EVOKE"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/don-t-reinvent-the-wheel-efficient","slug":"don-t-reinvent-the-wheel-efficient","title":"Don't Reinvent the Wheel: Efficient Instruction-Following Text Embedding based on Guided Space Transformation","date":"2025-05-30","arxiv_id":"2505.24754","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/don-t-reinvent-the-wheel-efficient#ran","syntology_url":"https://syntology.ai/paper/2505.24754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24754"}},"official":{"repos":["yingchaojiefeng/gstransform"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/let-them-talk-audio-driven-multi-person","slug":"let-them-talk-audio-driven-multi-person","title":"Let Them Talk: Audio-Driven Multi-Person Conversational Video Generation","date":"2025-05-28","arxiv_id":"2505.22647","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-them-talk-audio-driven-multi-person#ran","syntology_url":"https://syntology.ai/paper/2505.22647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22647"}},"official":{"repos":["meigen-ai/multitalk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/speech-ifeval-evaluating-instruction","slug":"speech-ifeval-evaluating-instruction","title":"Speech-IFEval: Evaluating Instruction-Following and Quantifying Catastrophic Forgetting in Speech-Aware Language Models","date":"2025-05-25","arxiv_id":"2505.19037","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/speech-ifeval-evaluating-instruction#ran","syntology_url":"https://syntology.ai/paper/2505.19037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19037"}},"official":{"repos":["kehanlu/speech-ifeval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sparse-activation-editing-for-reliable","slug":"sparse-activation-editing-for-reliable","title":"Sparse Activation Editing for Reliable Instruction Following in Narratives","date":"2025-05-22","arxiv_id":"2505.16505","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-activation-editing-for-reliable#ran","syntology_url":"https://syntology.ai/paper/2505.16505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16505"}},"official":null}},{"url":"/paper/lavida-a-large-diffusion-language-model-for","slug":"lavida-a-large-diffusion-language-model-for","title":"LaViDa: A Large Diffusion Language Model for Multimodal Understanding","date":"2025-05-22","arxiv_id":"2505.16839","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/lavida-a-large-diffusion-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2505.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16839"}},"official":{"repos":["jacklishufan/lavida"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/internal-causal-mechanisms-robustly-predict","slug":"internal-causal-mechanisms-robustly-predict","title":"Internal Causal Mechanisms Robustly Predict Language Model Out-of-Distribution Behaviors","date":"2025-05-17","arxiv_id":"2505.11770","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internal-causal-mechanisms-robustly-predict#ran","syntology_url":"https://syntology.ai/paper/2505.11770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11770"}},"official":{"repos":["explanare/ood-prediction"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/healthbench-evaluating-large-language-models","slug":"healthbench-evaluating-large-language-models","title":"HealthBench: Evaluating Large Language Models Towards Improved Human Health","date":"2025-05-13","arxiv_id":"2505.08775","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/healthbench-evaluating-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2505.08775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.08775"}},"official":{"repos":["openai/simple-evals"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-omni2-llm-based-real-time-spoken","slug":"llama-omni2-llm-based-real-time-spoken","title":"LLaMA-Omni2: LLM-based Real-time Spoken Chatbot with Autoregressive Streaming Speech Synthesis","date":"2025-05-05","arxiv_id":"2505.02625","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":3,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llama-omni2-llm-based-real-time-spoken#ran","syntology_url":"https://syntology.ai/paper/2505.02625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02625"}},"official":{"repos":["ictnlp/llama-omni2"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/realwebassist-a-benchmark-for-long-horizon","slug":"realwebassist-a-benchmark-for-long-horizon","title":"RealWebAssist: A Benchmark for Long-Horizon Web Assistance with Real-World Users","date":"2025-04-14","arxiv_id":"2504.10445","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/realwebassist-a-benchmark-for-long-horizon#ran","syntology_url":"https://syntology.ai/paper/2504.10445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10445"}},"official":{"repos":["scai-jhu/realwebassist"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-ifengine-towards-multimodal-instruction","slug":"mm-ifengine-towards-multimodal-instruction","title":"MM-IFEngine: Towards Multimodal Instruction Following","date":"2025-04-10","arxiv_id":"2504.07957","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mm-ifengine-towards-multimodal-instruction#ran","syntology_url":"https://syntology.ai/paper/2504.07957","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07957"}},"official":{"repos":["syuan03/mm-ifengine"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/crystalformer-rl-reinforcement-fine-tuning","slug":"crystalformer-rl-reinforcement-fine-tuning","title":"CrystalFormer-RL: Reinforcement Fine-Tuning for Materials Design","date":"2025-04-03","arxiv_id":"2504.02367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crystalformer-rl-reinforcement-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2504.02367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02367"}},"official":{"repos":["deepmodeling/crystalformer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vargpt-v1-1-improve-visual-autoregressive","slug":"vargpt-v1-1-improve-visual-autoregressive","title":"VARGPT-v1.1: Improve Visual Autoregressive Large Unified Model via Iterative Instruction Tuning and Reinforcement Learning","date":"2025-04-03","arxiv_id":"2504.02949","repositories_listed":2,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/vargpt-v1-1-improve-visual-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2504.02949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02949"}},"official":{"repos":["vargpt-family/vargpt-v1.1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/rec-r1-bridging-generative-large-language","slug":"rec-r1-bridging-generative-large-language","title":"Rec-R1: Bridging Generative Large Language Models and User-Centric Recommendation Systems via Reinforcement Learning","date":"2025-03-31","arxiv_id":"2503.24289","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rec-r1-bridging-generative-large-language#ran","syntology_url":"https://syntology.ai/paper/2503.24289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24289"}},"official":{"repos":["linjc16/Rec-R1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/insvie-1m-effective-instruction-based-video","slug":"insvie-1m-effective-instruction-based-video","title":"InsViE-1M: Effective Instruction-based Video Editing with Elaborate Dataset Construction","date":"2025-03-26","arxiv_id":"2503.20287","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":7,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/insvie-1m-effective-instruction-based-video#ran","syntology_url":"https://syntology.ai/paper/2503.20287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20287"}},"official":{"repos":["langmanbusi/insvie"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/simplerl-zoo-investigating-and-taming-zero","slug":"simplerl-zoo-investigating-and-taming-zero","title":"SimpleRL-Zoo: Investigating and Taming Zero Reinforcement Learning for Open Base Models in the Wild","date":"2025-03-24","arxiv_id":"2503.18892","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simplerl-zoo-investigating-and-taming-zero#ran","syntology_url":"https://syntology.ai/paper/2503.18892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18892"}},"official":null}},{"url":"/paper/does-context-matter-contextualjudgebench-for","slug":"does-context-matter-contextualjudgebench-for","title":"Does Context Matter? ContextualJudgeBench for Evaluating LLM-based Judges in Contextual Settings","date":"2025-03-19","arxiv_id":"2503.15620","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/does-context-matter-contextualjudgebench-for#ran","syntology_url":"https://syntology.ai/paper/2503.15620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15620"}},"official":{"repos":["salesforceairesearch/contextualjudgebench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-more-a-comparative-study-of-llms-and","slug":"llava-more-a-comparative-study-of-llms-and","title":"LLaVA-MORE: A Comparative Study of LLMs and Visual Backbones for Enhanced Visual Instruction Tuning","date":"2025-03-19","arxiv_id":"2503.15621","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/llava-more-a-comparative-study-of-llms-and#ran","syntology_url":"https://syntology.ai/paper/2503.15621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15621"}},"official":{"repos":["aimagelab/LLaVA-MORE"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/robust-multi-objective-controlled-decoding-of","slug":"robust-multi-objective-controlled-decoding-of","title":"Robust Multi-Objective Controlled Decoding of Large Language Models","date":"2025-03-11","arxiv_id":"2503.08796","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/robust-multi-objective-controlled-decoding-of#ran","syntology_url":"https://syntology.ai/paper/2503.08796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08796"}},"official":{"repos":["williambankes/robust-multi-objective-decoding"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xifbench-evaluating-large-language-models-on","slug":"xifbench-evaluating-large-language-models-on","title":"XIFBench: Evaluating Large Language Models on Multilingual Instruction Following","date":"2025-03-10","arxiv_id":"2503.07539","repositories_listed":0,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/xifbench-evaluating-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2503.07539","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07539"}},"official":null}},{"url":"/paper/seedream-2-0-a-native-chinese-english","slug":"seedream-2-0-a-native-chinese-english","title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","date":"2025-03-10","arxiv_id":"2503.07703","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/seedream-2-0-a-native-chinese-english#ran","syntology_url":"https://syntology.ai/paper/2503.07703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07703"}},"official":null}},{"url":"/paper/wildifeval-instruction-following-in-the-wild","slug":"wildifeval-instruction-following-in-the-wild","title":"WildIFEval: Instruction Following in the Wild","date":"2025-03-09","arxiv_id":"2503.06573","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wildifeval-instruction-following-in-the-wild#ran","syntology_url":"https://syntology.ai/paper/2503.06573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06573"}},"official":{"repos":["gililior/wild-if-eval-code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/routereval-a-comprehensive-benchmark-for","slug":"routereval-a-comprehensive-benchmark-for","title":"RouterEval: A Comprehensive Benchmark for Routing LLMs to Explore Model-level Scaling Up in LLMs","date":"2025-03-08","arxiv_id":"2503.10657","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/routereval-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2503.10657","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10657"}},"official":{"repos":["milkthink-lab/routereval"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/re-imagining-multimodal-instruction-tuning-a","slug":"re-imagining-multimodal-instruction-tuning-a","title":"Re-Imagining Multimodal Instruction Tuning: A Representation View","date":"2025-03-02","arxiv_id":"2503.00723","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/re-imagining-multimodal-instruction-tuning-a#ran","syntology_url":"https://syntology.ai/paper/2503.00723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00723"}},"official":{"repos":["comeandcode/MRT"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/agentic-reward-modeling-integrating-human","slug":"agentic-reward-modeling-integrating-human","title":"Agentic Reward Modeling: Integrating Human Preferences with Verifiable Correctness Signals for Reliable Reward Systems","date":"2025-02-26","arxiv_id":"2502.19328","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":3,"n_ran_checked":8,"n_instrument":4,"n_unverified":2,"n_honours":4,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"12 ran (of which 3 constructed an object rather than computing a result; 8 with no instrument failure: 4 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-reward-modeling-integrating-human#ran","syntology_url":"https://syntology.ai/paper/2502.19328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19328"}},"official":{"repos":["thu-keg/agentic-reward-modeling"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":3,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/stay-focused-problem-drift-in-multi-agent","slug":"stay-focused-problem-drift-in-multi-agent","title":"Stay Focused: Problem Drift in Multi-Agent Debate","date":"2025-02-26","arxiv_id":"2502.19559","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stay-focused-problem-drift-in-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2502.19559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19559"}},"official":{"repos":["jonas-becker/problem-drift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rank1-test-time-compute-for-reranking-in","slug":"rank1-test-time-compute-for-reranking-in","title":"Rank1: Test-Time Compute for Reranking in Information Retrieval","date":"2025-02-25","arxiv_id":"2502.18418","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rank1-test-time-compute-for-reranking-in#ran","syntology_url":"https://syntology.ai/paper/2502.18418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18418"}},"official":{"repos":["orionw/rank1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/textgames-learning-to-self-play-text-based","slug":"textgames-learning-to-self-play-text-based","title":"TextGames: Learning to Self-Play Text-Based Puzzle Games via Language Model Reasoning","date":"2025-02-25","arxiv_id":"2502.18431","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/textgames-learning-to-self-play-text-based#ran","syntology_url":"https://syntology.ai/paper/2502.18431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18431"}},"official":{"repos":["fhudi/textgames"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/sotopia-o-dynamic-strategy-injection-learning","slug":"sotopia-o-dynamic-strategy-injection-learning","title":"SOTOPIA-$Ω$: Dynamic Strategy Injection Learning and Social Instruction Following Evaluation for Social Agents","date":"2025-02-21","arxiv_id":"2502.15538","repositories_listed":1,"syntology":{"n":19,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":14,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":19,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/sotopia-o-dynamic-strategy-injection-learning#ran","syntology_url":"https://syntology.ai/paper/2502.15538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15538"}},"official":{"repos":["WYRipple/SOTOPIA-Omega"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/structflowbench-a-structured-flow-benchmark","slug":"structflowbench-a-structured-flow-benchmark","title":"StructFlowBench: A Structured Flow Benchmark for Multi-turn Instruction Following","date":"2025-02-20","arxiv_id":"2502.14494","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/structflowbench-a-structured-flow-benchmark#ran","syntology_url":"https://syntology.ai/paper/2502.14494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14494"}},"official":{"repos":["mlgroupjlu/structflowbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tess-2-a-large-scale-generalist-diffusion","slug":"tess-2-a-large-scale-generalist-diffusion","title":"TESS 2: A Large-Scale Generalist Diffusion Language Model","date":"2025-02-19","arxiv_id":"2502.13917","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tess-2-a-large-scale-generalist-diffusion#ran","syntology_url":"https://syntology.ai/paper/2502.13917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13917"}},"official":{"repos":["hamishivi/tess-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-diffusion-models","slug":"large-language-diffusion-models","title":"Large Language Diffusion Models","date":"2025-02-14","arxiv_id":"2502.09992","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-diffusion-models#ran","syntology_url":"https://syntology.ai/paper/2502.09992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09992"}},"official":null}},{"url":"/paper/iheval-evaluating-language-models-on","slug":"iheval-evaluating-language-models-on","title":"IHEval: Evaluating Language Models on Following the Instruction Hierarchy","date":"2025-02-12","arxiv_id":"2502.08745","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iheval-evaluating-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2502.08745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.08745"}},"official":{"repos":["ytyz1307zzh/IHEval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ultraif-advancing-instruction-following-from","slug":"ultraif-advancing-instruction-following-from","title":"UltraIF: Advancing Instruction Following from the Wild","date":"2025-02-06","arxiv_id":"2502.04153","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ultraif-advancing-instruction-following-from#ran","syntology_url":"https://syntology.ai/paper/2502.04153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04153"}},"official":{"repos":["kkk-an/ultraif"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/code-blockwise-control-for-denoising","slug":"code-blockwise-control-for-denoising","title":"CoDe: Blockwise Control for Denoising Diffusion Models","date":"2025-02-03","arxiv_id":"2502.00968","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code-blockwise-control-for-denoising#ran","syntology_url":"https://syntology.ai/paper/2502.00968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.00968"}},"official":{"repos":["anujinho/code"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/critique-fine-tuning-learning-to-critique-is","slug":"critique-fine-tuning-learning-to-critique-is","title":"Critique Fine-Tuning: Learning to Critique is More Effective than Learning to Imitate","date":"2025-01-29","arxiv_id":"2501.17703","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/critique-fine-tuning-learning-to-critique-is#ran","syntology_url":"https://syntology.ai/paper/2501.17703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.17703"}},"official":null}},{"url":"/paper/curiosity-driven-reinforcement-learning-from","slug":"curiosity-driven-reinforcement-learning-from","title":"Curiosity-Driven Reinforcement Learning from Human Feedback","date":"2025-01-20","arxiv_id":"2501.11463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/curiosity-driven-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2501.11463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11463"}},"official":{"repos":["ernie-research/cd-rlhf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/demystifying-domain-adaptive-post-training","slug":"demystifying-domain-adaptive-post-training","title":"Demystifying Domain-adaptive Post-training for Financial LLMs","date":"2025-01-09","arxiv_id":"2501.04961","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/demystifying-domain-adaptive-post-training#ran","syntology_url":"https://syntology.ai/paper/2501.04961","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04961"}},"official":{"repos":["salesforceairesearch/findap"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/find-the-intention-of-instruction","slug":"find-the-intention-of-instruction","title":"Find the Intention of Instruction: Comprehensive Evaluation of Instruction Understanding for Large Language Models","date":"2024-12-27","arxiv_id":"2412.19450","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/find-the-intention-of-instruction#ran","syntology_url":"https://syntology.ai/paper/2412.19450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.19450"}},"official":{"repos":["hyeonseokk/ioinst"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/align-anything-training-all-modality-models","slug":"align-anything-training-all-modality-models","title":"Align Anything: Training All-Modality Models to Follow Instructions with Language Feedback","date":"2024-12-20","arxiv_id":"2412.15838","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/align-anything-training-all-modality-models#ran","syntology_url":"https://syntology.ai/paper/2412.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15838"}},"official":{"repos":["pku-alignment/align-anything"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/qwen2-5-technical-report","slug":"qwen2-5-technical-report","title":"Qwen2.5 Technical Report","date":"2024-12-19","arxiv_id":"2412.15115","repositories_listed":6,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/qwen2-5-technical-report#ran","syntology_url":"https://syntology.ai/paper/2412.15115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15115"}},"official":{"repos":["qwenlm/qwen2.5"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/spar-self-play-with-tree-search-refinement-to","slug":"spar-self-play-with-tree-search-refinement-to","title":"SPaR: Self-Play with Tree-Search Refinement to Improve Instruction-Following in Large Language Models","date":"2024-12-16","arxiv_id":"2412.11605","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/spar-self-play-with-tree-search-refinement-to#ran","syntology_url":"https://syntology.ai/paper/2412.11605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11605"}},"official":{"repos":["thu-coai/spar"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-instruction-tuning-with-500x-fewer","slug":"visual-instruction-tuning-with-500x-fewer","title":"LLaVA Steering: Visual Instruction Tuning with 500x Fewer Parameters through Modality Linear Representation-Steering","date":"2024-12-16","arxiv_id":"2412.12359","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/visual-instruction-tuning-with-500x-fewer#ran","syntology_url":"https://syntology.ai/paper/2412.12359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12359"}},"official":{"repos":["bibisbar/LLaVA-Steering"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sloth-scaling-laws-for-llm-skills-to-predict","slug":"sloth-scaling-laws-for-llm-skills-to-predict","title":"Sloth: scaling laws for LLM skills to predict multi-benchmark performance across families","date":"2024-12-09","arxiv_id":"2412.06540","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sloth-scaling-laws-for-llm-skills-to-predict#ran","syntology_url":"https://syntology.ai/paper/2412.06540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06540"}},"official":{"repos":["felipemaiapolo/sloth"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kasa-knowledge-aware-singular-value","slug":"kasa-knowledge-aware-singular-value","title":"KaSA: Knowledge-Aware Singular-Value Adaptation of Large Language Models","date":"2024-12-08","arxiv_id":"2412.06071","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/kasa-knowledge-aware-singular-value#ran","syntology_url":"https://syntology.ai/paper/2412.06071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06071"}},"official":{"repos":["juyongjiang/kasa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/prefixkv-adaptive-prefix-kv-cache-is-what","slug":"prefixkv-adaptive-prefix-kv-cache-is-what","title":"PrefixKV: Adaptive Prefix KV Cache is What Vision Instruction-Following Models Need for Efficient Generation","date":"2024-12-04","arxiv_id":"2412.03409","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prefixkv-adaptive-prefix-kv-cache-is-what#ran","syntology_url":"https://syntology.ai/paper/2412.03409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03409"}},"official":{"repos":["THU-MIG/PrefixKV"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/agri-llava-knowledge-infused-large-multimodal","slug":"agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","arxiv_id":"2412.02158","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agri-llava-knowledge-infused-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2412.02158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02158"}},"official":{"repos":["kki2eve/agri-llava"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/showui-one-vision-language-action-model-for","slug":"showui-one-vision-language-action-model-for","title":"ShowUI: One Vision-Language-Action Model for GUI Visual Agent","date":"2024-11-26","arxiv_id":"2411.17465","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/showui-one-vision-language-action-model-for#ran","syntology_url":"https://syntology.ai/paper/2411.17465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17465"}},"official":{"repos":["showlab/showui"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-instruction-tuning-an","slug":"parameter-efficient-instruction-tuning-an","title":"Parameter Efficient Instruction Tuning: An Empirical Study","date":"2024-11-25","arxiv_id":"2411.16775","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parameter-efficient-instruction-tuning-an#ran","syntology_url":"https://syntology.ai/paper/2411.16775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16775"}},"official":{"repos":["AdaBit-AI/parameter_efficient_instruction_tuning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lhrs-bot-nova-improved-multimodal-large","slug":"lhrs-bot-nova-improved-multimodal-large","title":"LHRS-Bot-Nova: Improved Multimodal Large Language Model for Remote Sensing Vision-Language Interpretation","date":"2024-11-14","arxiv_id":"2411.09301","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lhrs-bot-nova-improved-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2411.09301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.09301"}},"official":{"repos":["NJU-LHRS/LHRS-Bot"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lifbench-evaluating-the-instruction-following","slug":"lifbench-evaluating-the-instruction-following","title":"LIFBench: Evaluating the Instruction Following Performance and Stability of Large Language Models in Long-Context Scenarios","date":"2024-11-11","arxiv_id":"2411.07037","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lifbench-evaluating-the-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2411.07037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07037"}},"official":{"repos":["sheldonwu0327/lif-bench-2024"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/setlexsem-challenge-using-set-operations-to","slug":"setlexsem-challenge-using-set-operations-to","title":"SetLexSem Challenge: Using Set Operations to Evaluate the Lexical and Semantic Robustness of Language Models","date":"2024-11-11","arxiv_id":"2411.07336","repositories_listed":1,"syntology":{"n":23,"n_ran":19,"n_constructed":0,"n_ran_checked":16,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/setlexsem-challenge-using-set-operations-to#ran","syntology_url":"https://syntology.ai/paper/2411.07336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07336"}},"official":{"repos":["amazon-science/setlexsem-challenge"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-calibration-of-win-rate-estimation","slug":"bayesian-calibration-of-win-rate-estimation","title":"Bayesian Calibration of Win Rate Estimation with LLM Evaluators","date":"2024-11-07","arxiv_id":"2411.04424","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayesian-calibration-of-win-rate-estimation#ran","syntology_url":"https://syntology.ai/paper/2411.04424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04424"}},"official":{"repos":["yale-nlp/bay-calibration-llm-evaluators"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-loss-of-context-awareness-in-general","slug":"on-the-loss-of-context-awareness-in-general","title":"On the Loss of Context-awareness in General Instruction Fine-tuning","date":"2024-11-05","arxiv_id":"2411.02688","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-loss-of-context-awareness-in-general#ran","syntology_url":"https://syntology.ai/paper/2411.02688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02688"}},"official":{"repos":["YihanWang617/context_awareness"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/constraint-back-translation-improves-complex","slug":"constraint-back-translation-improves-complex","title":"Constraint Back-translation Improves Complex Instruction Following of Large Language Models","date":"2024-10-31","arxiv_id":"2410.24175","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/constraint-back-translation-improves-complex#ran","syntology_url":"https://syntology.ai/paper/2410.24175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24175"}},"official":{"repos":["thu-keg/crab"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llamo-large-language-model-based-molecular","slug":"llamo-large-language-model-based-molecular","title":"LLaMo: Large Language Model-based Molecular Graph Assistant","date":"2024-10-31","arxiv_id":"2411.00871","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llamo-large-language-model-based-molecular#ran","syntology_url":"https://syntology.ai/paper/2411.00871","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00871"}},"official":{"repos":["mlvlab/llamo"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-model-control-improving-multiple-large","slug":"cross-model-control-improving-multiple-large","title":"Cross-model Control: Improving Multiple Large Language Models in One-time Training","date":"2024-10-23","arxiv_id":"2410.17599","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cross-model-control-improving-multiple-large#ran","syntology_url":"https://syntology.ai/paper/2410.17599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17599"}},"official":{"repos":["wujwyi/cmc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-lingual-transfer-of-reward-models-in","slug":"cross-lingual-transfer-of-reward-models-in","title":"Cross-lingual Transfer of Reward Models in Multilingual Alignment","date":"2024-10-23","arxiv_id":"2410.18027","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-lingual-transfer-of-reward-models-in#ran","syntology_url":"https://syntology.ai/paper/2410.18027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18027"}},"official":{"repos":["iq-kaist/rm-lingual-transfer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-if-benchmarking-llms-on-multi-turn-and","slug":"multi-if-benchmarking-llms-on-multi-turn-and","title":"Multi-IF: Benchmarking LLMs on Multi-Turn and Multilingual Instructions Following","date":"2024-10-21","arxiv_id":"2410.15553","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-if-benchmarking-llms-on-multi-turn-and#ran","syntology_url":"https://syntology.ai/paper/2410.15553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15553"}},"official":{"repos":["facebookresearch/Multi-IF"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/logu-long-form-generation-with-uncertainty","slug":"logu-long-form-generation-with-uncertainty","title":"LoGU: Long-form Generation with Uncertainty Expressions","date":"2024-10-18","arxiv_id":"2410.14309","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/logu-long-form-generation-with-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2410.14309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14309"}},"official":{"repos":["rhyang2021/logu"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-llms-know-internally-when-they-follow","slug":"do-llms-know-internally-when-they-follow","title":"Do LLMs \"know\" internally when they follow instructions?","date":"2024-10-18","arxiv_id":"2410.14516","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/do-llms-know-internally-when-they-follow#ran","syntology_url":"https://syntology.ai/paper/2410.14516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14516"}},"official":{"repos":["apple/ml-internal-llms-instruction-following"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/loldu-low-rank-adaptation-via-lower-diag","slug":"loldu-low-rank-adaptation-via-lower-diag","title":"LoLDU: Low-Rank Adaptation via Lower-Diag-Upper Decomposition for Parameter-Efficient Fine-Tuning","date":"2024-10-17","arxiv_id":"2410.13618","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/loldu-low-rank-adaptation-via-lower-diag#ran","syntology_url":"https://syntology.ai/paper/2410.13618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13618"}},"official":{"repos":["skddj/loldu"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-chunking-learning-efficient-text","slug":"meta-chunking-learning-efficient-text","title":"Meta-Chunking: Learning Text Segmentation and Semantic Completion via Logical Perception","date":"2024-10-16","arxiv_id":"2410.12788","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-chunking-learning-efficient-text#ran","syntology_url":"https://syntology.ai/paper/2410.12788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12788"}},"official":{"repos":["IAAR-Shanghai/Meta-Chunking"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-instruction-following-in-language-1","slug":"improving-instruction-following-in-language-1","title":"Improving Instruction-Following in Language Models through Activation Steering","date":"2024-10-15","arxiv_id":"2410.12877","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-instruction-following-in-language-1#ran","syntology_url":"https://syntology.ai/paper/2410.12877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12877"}},"official":null}},{"url":"/paper/how-to-leverage-demonstration-data-in","slug":"how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10093","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-leverage-demonstration-data-in#ran","syntology_url":"https://syntology.ai/paper/2410.10093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10093"}},"official":{"repos":["tengxiao1/gsil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-general-instruction-following","slug":"toward-general-instruction-following","title":"Toward General Instruction-Following Alignment for Retrieval-Augmented Generation","date":"2024-10-12","arxiv_id":"2410.09584","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/toward-general-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2410.09584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09584"}},"official":{"repos":["dongguanting/FollowRAG"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-imbalance-driven-rewarding-for","slug":"language-imbalance-driven-rewarding-for","title":"Language Imbalance Driven Rewarding for Multilingual Self-improving","date":"2024-10-11","arxiv_id":"2410.08964","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/language-imbalance-driven-rewarding-for#ran","syntology_url":"https://syntology.ai/paper/2410.08964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08964"}},"official":{"repos":["znlp/language-imbalance-driven-rewarding"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/copesd-a-multi-level-surgical-motion-dataset","slug":"copesd-a-multi-level-surgical-motion-dataset","title":"CoPESD: A Multi-Level Surgical Motion Dataset for Training Large Vision-Language Models to Co-Pilot Endoscopic Submucosal Dissection","date":"2024-10-10","arxiv_id":"2410.07540","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/copesd-a-multi-level-surgical-motion-dataset#ran","syntology_url":"https://syntology.ai/paper/2410.07540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07540"}},"official":{"repos":["gkw0010/copesd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-augmented-data-enhances-direct","slug":"reward-augmented-data-enhances-direct","title":"Reward-Augmented Data Enhances Direct Preference Alignment of LLMs","date":"2024-10-10","arxiv_id":"2410.08067","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-augmented-data-enhances-direct#ran","syntology_url":"https://syntology.ai/paper/2410.08067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08067"}},"official":{"repos":["shenao-zhang/reward-augmented-preference"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aria-an-open-multimodal-native-mixture-of","slug":"aria-an-open-multimodal-native-mixture-of","title":"Aria: An Open Multimodal Native Mixture-of-Experts Model","date":"2024-10-08","arxiv_id":"2410.05993","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aria-an-open-multimodal-native-mixture-of#ran","syntology_url":"https://syntology.ai/paper/2410.05993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05993"}},"official":{"repos":["rhymes-ai/aria"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/teochat-a-large-vision-language-assistant-for","slug":"teochat-a-large-vision-language-assistant-for","title":"TEOChat: A Large Vision-Language Assistant for Temporal Earth Observation Data","date":"2024-10-08","arxiv_id":"2410.06234","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/teochat-a-large-vision-language-assistant-for#ran","syntology_url":"https://syntology.ai/paper/2410.06234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06234"}},"official":{"repos":["ermongroup/teochat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/cs4-measuring-the-creativity-of-large","slug":"cs4-measuring-the-creativity-of-large","title":"CS4: Measuring the Creativity of Large Language Models Automatically by Controlling the Number of Story-Writing Constraints","date":"2024-10-05","arxiv_id":"2410.04197","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cs4-measuring-the-creativity-of-large#ran","syntology_url":"https://syntology.ai/paper/2410.04197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04197"}},"official":{"repos":["anirudhlakkaraju/cs4_benchmark"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/commonit-commonality-aware-instruction-tuning","slug":"commonit-commonality-aware-instruction-tuning","title":"CommonIT: Commonality-Aware Instruction Tuning for Large Language Models via Data Partitions","date":"2024-10-04","arxiv_id":"2410.03077","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/commonit-commonality-aware-instruction-tuning#ran","syntology_url":"https://syntology.ai/paper/2410.03077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03077"}},"official":{"repos":["raojay7/commonit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/robin3d-improving-3d-large-language-model-via","slug":"robin3d-improving-3d-large-language-model-via","title":"Robin3D: Improving 3D Large Language Model via Robust Instruction Tuning","date":"2024-09-30","arxiv_id":"2410.00255","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robin3d-improving-3d-large-language-model-via#ran","syntology_url":"https://syntology.ai/paper/2410.00255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00255"}},"official":{"repos":["weitaikang/robin3d"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/infer-human-s-intentions-before-following","slug":"infer-human-s-intentions-before-following","title":"Infer Human's Intentions Before Following Natural Language Instructions","date":"2024-09-26","arxiv_id":"2409.18073","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infer-human-s-intentions-before-following#ran","syntology_url":"https://syntology.ai/paper/2409.18073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18073"}},"official":{"repos":["simon-wan/fiser"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/eventhallusion-diagnosing-event","slug":"eventhallusion-diagnosing-event","title":"EventHallusion: Diagnosing Event Hallucinations in Video LLMs","date":"2024-09-25","arxiv_id":"2409.16597","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eventhallusion-diagnosing-event#ran","syntology_url":"https://syntology.ai/paper/2409.16597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16597"}},"official":{"repos":["stevetich/eventhallusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mitigating-the-bias-of-large-language-model","slug":"mitigating-the-bias-of-large-language-model","title":"Mitigating the Bias of Large Language Model Evaluation","date":"2024-09-25","arxiv_id":"2409.16788","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-the-bias-of-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2409.16788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16788"}},"official":{"repos":["Joe-Hall-Lee/Debias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toolplanner-a-tool-augmented-llm-for-multi","slug":"toolplanner-a-tool-augmented-llm-for-multi","title":"ToolPlanner: A Tool Augmented LLM for Multi Granularity Instructions with Path Planning and Feedback","date":"2024-09-23","arxiv_id":"2409.14826","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/toolplanner-a-tool-augmented-llm-for-multi#ran","syntology_url":"https://syntology.ai/paper/2409.14826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14826"}},"official":{"repos":["xiaomi/toolplanner"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/archon-an-architecture-search-framework-for","slug":"archon-an-architecture-search-framework-for","title":"Archon: An Architecture Search Framework for Inference-Time Techniques","date":"2024-09-23","arxiv_id":"2409.15254","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/archon-an-architecture-search-framework-for#ran","syntology_url":"https://syntology.ai/paper/2409.15254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15254"}},"official":{"repos":["scalingintelligence/archon"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2409-13989","slug":"2409-13989","title":"ChemEval: A Comprehensive Multi-Level Chemical Evaluation for Large Language Models","date":"2024-09-21","arxiv_id":"2409.13989","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":18,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2409-13989#ran","syntology_url":"https://syntology.ai/paper/2409.13989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13989"}},"official":{"repos":["ustc-starteam/chemeval"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spinning-the-golden-thread-benchmarking-long","slug":"spinning-the-golden-thread-benchmarking-long","title":"LongGenBench: Benchmarking Long-Form Generation in Long Context LLMs","date":"2024-09-03","arxiv_id":"2409.02076","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spinning-the-golden-thread-benchmarking-long#ran","syntology_url":"https://syntology.ai/paper/2409.02076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02076"}},"official":{"repos":["mozhu621/SGT","mozhu621/longgenbench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-judge-selective-instruction-following","slug":"self-judge-selective-instruction-following","title":"Self-Judge: Selective Instruction Following with Alignment Self-Evaluation","date":"2024-09-02","arxiv_id":"2409.00935","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-judge-selective-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2409.00935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00935"}},"official":{"repos":["nusnlp/Self-J"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genagent-build-collaborative-ai-systems-with","slug":"genagent-build-collaborative-ai-systems-with","title":"ComfyBench: Benchmarking LLM-based Agents in ComfyUI for Autonomously Designing Collaborative AI Systems","date":"2024-09-02","arxiv_id":"2409.01392","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genagent-build-collaborative-ai-systems-with#ran","syntology_url":"https://syntology.ai/paper/2409.01392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01392"}},"official":{"repos":["xxyQwQ/ComfyBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scilitllm-how-to-adapt-llms-for-scientific","slug":"scilitllm-how-to-adapt-llms-for-scientific","title":"SciLitLLM: How to Adapt LLMs for Scientific Literature Understanding","date":"2024-08-28","arxiv_id":"2408.15545","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scilitllm-how-to-adapt-llms-for-scientific#ran","syntology_url":"https://syntology.ai/paper/2408.15545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15545"}},"official":{"repos":["dptech-corp/Uni-SMART"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instruct-skillmix-a-powerful-pipeline-for-llm","slug":"instruct-skillmix-a-powerful-pipeline-for-llm","title":"Instruct-SkillMix: A Powerful Pipeline for LLM Instruction Tuning","date":"2024-08-27","arxiv_id":"2408.14774","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instruct-skillmix-a-powerful-pipeline-for-llm#ran","syntology_url":"https://syntology.ai/paper/2408.14774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14774"}},"official":{"repos":["princeton-pli/Instruct-SkillMix"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llms-are-biased-towards-output-formats","slug":"llms-are-biased-towards-output-formats","title":"LLMs Are Biased Towards Output Formats! Systematically Evaluating and Mitigating Output Format Bias of LLMs","date":"2024-08-16","arxiv_id":"2408.08656","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llms-are-biased-towards-output-formats#ran","syntology_url":"https://syntology.ai/paper/2408.08656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08656"}},"official":{"repos":["dxlong2000/FormatEval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-and-modeling-correlations-in","slug":"bridging-and-modeling-correlations-in","title":"Bridging and Modeling Correlations in Pairwise Data for Direct Preference Optimization","date":"2024-08-14","arxiv_id":"2408.07471","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-and-modeling-correlations-in#ran","syntology_url":"https://syntology.ai/paper/2408.07471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07471"}},"official":{"repos":["YJiangcm/BMC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/1-5-pints-technical-report-pretraining-in","slug":"1-5-pints-technical-report-pretraining-in","title":"1.5-Pints Technical Report: Pretraining in Days, Not Months -- Your Language Model Thrives on Quality Data","date":"2024-08-07","arxiv_id":"2408.03506","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/1-5-pints-technical-report-pretraining-in#ran","syntology_url":"https://syntology.ai/paper/2408.03506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03506"}},"official":{"repos":["Pints-AI/1.5-Pints"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/extend-model-merging-from-fine-tuned-to-pre","slug":"extend-model-merging-from-fine-tuned-to-pre","title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","date":"2024-08-06","arxiv_id":"2408.03092","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/extend-model-merging-from-fine-tuned-to-pre#ran","syntology_url":"https://syntology.ai/paper/2408.03092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03092"}},"official":{"repos":["yule-BUAA/MergeLLM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seallms-3-open-foundation-and-chat","slug":"seallms-3-open-foundation-and-chat","title":"SeaLLMs 3: Open Foundation and Chat Multilingual Large Language Models for Southeast Asian Languages","date":"2024-07-29","arxiv_id":"2407.19672","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seallms-3-open-foundation-and-chat#ran","syntology_url":"https://syntology.ai/paper/2407.19672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19672"}},"official":{"repos":["DAMO-NLP-SG/SeaExam"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/realfred-an-embodied-instruction-following","slug":"realfred-an-embodied-instruction-following","title":"ReALFRED: An Embodied Instruction Following Benchmark in Photo-Realistic Environments","date":"2024-07-26","arxiv_id":"2407.18550","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/realfred-an-embodied-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2407.18550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18550"}},"official":{"repos":["snumprlab/realfred"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"d93b821930aea337c1c2ee8519e2f477cd015c4a775040127d8bc538e03b5db0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}