{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/instruction-following/papers/4","list_of":"/task/instruction-following","task":"Instruction Following","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":12,"rows_per_page":100,"rows":[301,400],"of":1135,"counts":{"archive_papers_tagged":1135,"with_a_code_link":609,"where_syntology_ran_a_sample":311,"not_listed_spam_title":0,"listed":1135,"listed_where_code_ran":311,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":255,"every_run_a_failure_of_syntologys_instrument":56,"listed_with_a_run_with_no_instrument_failure":255,"listed_every_run_a_failure_of_syntologys_instrument":56,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/instruction-following","prev":"/task/instruction-following/papers/3","next":"/task/instruction-following/papers/5","papers":[{"url":"/paper/llava-vsd-large-language-and-vision-assistant","slug":"llava-vsd-large-language-and-vision-assistant","title":"LLaVA-VSD: Large Language-and-Vision Assistant for Visual Spatial Description","date":"2024-08-09","arxiv_id":"2408.04957","repositories_listed":1,"syntology":null},{"url":"/paper/1-5-pints-technical-report-pretraining-in","slug":"1-5-pints-technical-report-pretraining-in","title":"1.5-Pints Technical Report: Pretraining in Days, Not Months -- Your Language Model Thrives on Quality Data","date":"2024-08-07","arxiv_id":"2408.03506","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/1-5-pints-technical-report-pretraining-in#ran","syntology_url":"https://syntology.ai/paper/2408.03506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03506"}},"official":{"repos":["Pints-AI/1.5-Pints"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/walledeval-a-comprehensive-safety-evaluation","slug":"walledeval-a-comprehensive-safety-evaluation","title":"WalledEval: A Comprehensive Safety Evaluation Toolkit for Large Language Models","date":"2024-08-07","arxiv_id":"2408.03837","repositories_listed":1,"syntology":null},{"url":"/paper/extend-model-merging-from-fine-tuned-to-pre","slug":"extend-model-merging-from-fine-tuned-to-pre","title":"Extend Model Merging from Fine-Tuned to Pre-Trained Large Language Models via Weight Disentanglement","date":"2024-08-06","arxiv_id":"2408.03092","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/extend-model-merging-from-fine-tuned-to-pre#ran","syntology_url":"https://syntology.ai/paper/2408.03092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03092"}},"official":{"repos":["yule-BUAA/MergeLLM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21417","slug":"2407-21417","title":"Dancing in Chains: Reconciling Instruction Following and Faithfulness in Language Models","date":"2024-07-31","arxiv_id":"2407.21417","repositories_listed":1,"syntology":null},{"url":"/paper/autonomous-improvement-of-instruction","slug":"autonomous-improvement-of-instruction","title":"Autonomous Improvement of Instruction Following Skills via Foundation Models","date":"2024-07-30","arxiv_id":"2407.20635","repositories_listed":1,"syntology":null},{"url":"/paper/realfred-an-embodied-instruction-following","slug":"realfred-an-embodied-instruction-following","title":"ReALFRED: An Embodied Instruction Following Benchmark in Photo-Realistic Environments","date":"2024-07-26","arxiv_id":"2407.18550","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/realfred-an-embodied-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2407.18550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18550"}},"official":{"repos":["snumprlab/realfred"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-inference-of-vision-instruction","slug":"efficient-inference-of-vision-instruction","title":"Efficient Inference of Vision Instruction-Following Models with Elastic Cache","date":"2024-07-25","arxiv_id":"2407.18121","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-inference-of-vision-instruction#ran","syntology_url":"https://syntology.ai/paper/2407.18121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18121"}},"official":{"repos":["liuzuyan/elasticcache"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/primeguard-safe-and-helpful-llms-through","slug":"primeguard-safe-and-helpful-llms-through","title":"PrimeGuard: Safe and Helpful LLMs through Tuning-Free Routing","date":"2024-07-23","arxiv_id":"2407.16318","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/primeguard-safe-and-helpful-llms-through#ran","syntology_url":"https://syntology.ai/paper/2407.16318","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16318"}},"official":{"repos":["dynamofl/primeguard"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/disco-embodied-navigation-and-interaction-via","slug":"disco-embodied-navigation-and-interaction-via","title":"DISCO: Embodied Navigation and Interaction via Differentiable Scene Semantics and Dual-level Control","date":"2024-07-20","arxiv_id":"2407.14758","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/disco-embodied-navigation-and-interaction-via#ran","syntology_url":"https://syntology.ai/paper/2407.14758","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14758"}},"official":{"repos":["allenxuuu/disco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/earthmarker-a-visual-prompt-learning","slug":"earthmarker-a-visual-prompt-learning","title":"EarthMarker: A Visual Prompting Multi-modal Large Language Model for Remote Sensing","date":"2024-07-18","arxiv_id":"2407.13596","repositories_listed":1,"syntology":null},{"url":"/paper/navgpt-2-unleashing-navigational-reasoning","slug":"navgpt-2-unleashing-navigational-reasoning","title":"NavGPT-2: Unleashing Navigational Reasoning Capability for Large Vision-Language Models","date":"2024-07-17","arxiv_id":"2407.12366","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/navgpt-2-unleashing-navigational-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.12366","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12366"}},"official":{"repos":["gengzezhou/navgpt-2"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-guide-better-task-specific-instruction","slug":"self-guide-better-task-specific-instruction","title":"SELF-GUIDE: Better Task-Specific Instruction Following via Self-Synthetic Finetuning","date":"2024-07-16","arxiv_id":"2407.12874","repositories_listed":1,"syntology":null},{"url":"/paper/farsinstruct-empowering-large-language-models","slug":"farsinstruct-empowering-large-language-models","title":"Empowering Persian LLMs for Instruction Following: A Novel Dataset and Training Approach","date":"2024-07-15","arxiv_id":"2407.11186","repositories_listed":1,"syntology":null},{"url":"/paper/instruction-following-with-goal-conditioned","slug":"instruction-following-with-goal-conditioned","title":"Instruction Following with Goal-Conditioned Reinforcement Learning in Virtual Environments","date":"2024-07-12","arxiv_id":"2407.09287","repositories_listed":1,"syntology":null},{"url":"/paper/lions-an-empirically-optimized-approach-to","slug":"lions-an-empirically-optimized-approach-to","title":"LIONs: An Empirically Optimized Approach to Align Language Models","date":"2024-07-09","arxiv_id":"2407.06542","repositories_listed":1,"syntology":null},{"url":"/paper/from-loops-to-oops-fallback-behaviors-of","slug":"from-loops-to-oops-fallback-behaviors-of","title":"From Loops to Oops: Fallback Behaviors of Language Models Under Uncertainty","date":"2024-07-08","arxiv_id":"2407.06071","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-loops-to-oops-fallback-behaviors-of#ran","syntology_url":"https://syntology.ai/paper/2407.06071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06071"}},"official":{"repos":["mivg/fallbacks"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mmsci-a-multimodal-multi-discipline-dataset","slug":"mmsci-a-multimodal-multi-discipline-dataset","title":"MMSci: A Dataset for Graduate-Level Multi-Discipline Multimodal Scientific Understanding","date":"2024-07-06","arxiv_id":"2407.04903","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mmsci-a-multimodal-multi-discipline-dataset#ran","syntology_url":"https://syntology.ai/paper/2407.04903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04903"}},"official":{"repos":["leezekun/mmsci"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/me-myself-and-ai-the-situational-awareness","slug":"me-myself-and-ai-the-situational-awareness","title":"Me, Myself, and AI: The Situational Awareness Dataset (SAD) for LLMs","date":"2024-07-05","arxiv_id":"2407.04694","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/me-myself-and-ai-the-situational-awareness#ran","syntology_url":"https://syntology.ai/paper/2407.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04694"}},"official":{"repos":["lrudl/sad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-complex-instruction-following","slug":"benchmarking-complex-instruction-following","title":"Benchmarking Complex Instruction-Following with Multiple Constraints Composition","date":"2024-07-04","arxiv_id":"2407.03978","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-complex-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2407.03978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03978"}},"official":{"repos":["thu-coai/complexbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/semantic-graphs-for-syntactic-simplification","slug":"semantic-graphs-for-syntactic-simplification","title":"Semantic Graphs for Syntactic Simplification: A Revisit from the Age of LLM","date":"2024-07-04","arxiv_id":"2407.04067","repositories_listed":1,"syntology":null},{"url":"/paper/mia-bench-towards-better-instruction","slug":"mia-bench-towards-better-instruction","title":"MIA-Bench: Towards Better Instruction Following Evaluation of Multimodal LLMs","date":"2024-07-01","arxiv_id":"2407.01509","repositories_listed":1,"syntology":null},{"url":"/paper/the-sifo-benchmark-investigating-the","slug":"the-sifo-benchmark-investigating-the","title":"The SIFo Benchmark: Investigating the Sequential Instruction Following Ability of Large Language Models","date":"2024-06-28","arxiv_id":"2406.19999","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-sifo-benchmark-investigating-the#ran","syntology_url":"https://syntology.ai/paper/2406.19999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19999"}},"official":{"repos":["shin-ee-chen/SIFo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/livebench-a-challenging-contamination-free","slug":"livebench-a-challenging-contamination-free","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","date":"2024-06-27","arxiv_id":"2406.19314","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/livebench-a-challenging-contamination-free#ran","syntology_url":"https://syntology.ai/paper/2406.19314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19314"}},"official":{"repos":["livebench/livebench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/suri-multi-constraint-instruction-following","slug":"suri-multi-constraint-instruction-following","title":"Suri: Multi-constraint Instruction Following for Long-form Text Generation","date":"2024-06-27","arxiv_id":"2406.19371","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/suri-multi-constraint-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2406.19371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19371"}},"official":{"repos":["chtmp223/suri"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dual-space-knowledge-distillation-for-large","slug":"dual-space-knowledge-distillation-for-large","title":"Dual-Space Knowledge Distillation for Large Language Models","date":"2024-06-25","arxiv_id":"2406.17328","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/dual-space-knowledge-distillation-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.17328","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17328"}},"official":{"repos":["songmzhang/dskd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autodetect-towards-a-unified-framework-for","slug":"autodetect-towards-a-unified-framework-for","title":"AutoDetect: Towards a Unified Framework for Automated Weakness Detection in Large Language Models","date":"2024-06-24","arxiv_id":"2406.16714","repositories_listed":1,"syntology":null},{"url":"/paper/lottery-ticket-adaptation-mitigating","slug":"lottery-ticket-adaptation-mitigating","title":"Lottery Ticket Adaptation: Mitigating Destructive Interference in LLMs","date":"2024-06-24","arxiv_id":"2406.16797","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lottery-ticket-adaptation-mitigating#ran","syntology_url":"https://syntology.ai/paper/2406.16797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16797"}},"official":{"repos":["kiddyboots216/lottery-ticket-adaptation"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/res-q-evaluating-code-editing-large-language","slug":"res-q-evaluating-code-editing-large-language","title":"RES-Q: Evaluating Code-Editing Large Language Model Systems at the Repository Scale","date":"2024-06-24","arxiv_id":"2406.16801","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/res-q-evaluating-code-editing-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.16801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16801"}},"official":{"repos":["qurrent-ai/res-q"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audiobench-a-universal-benchmark-for-audio","slug":"audiobench-a-universal-benchmark-for-audio","title":"AudioBench: A Universal Benchmark for Audio Large Language Models","date":"2024-06-23","arxiv_id":"2406.16020","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiobench-a-universal-benchmark-for-audio#ran","syntology_url":"https://syntology.ai/paper/2406.16020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16020"}},"official":{"repos":["audiollms/audiobench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-alignment-training-for-large-language","slug":"hybrid-alignment-training-for-large-language","title":"Hybrid Alignment Training for Large Language Models","date":"2024-06-21","arxiv_id":"2406.15178","repositories_listed":1,"syntology":null},{"url":"/paper/iwisdm-assessing-instruction-following-in","slug":"iwisdm-assessing-instruction-following-in","title":"IWISDM: Assessing instruction following in multimodal models at scale","date":"2024-06-20","arxiv_id":"2406.14343","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iwisdm-assessing-instruction-following-in#ran","syntology_url":"https://syntology.ai/paper/2406.14343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14343"}},"official":{"repos":["bashivanlab/iwisdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llasa-large-multimodal-agent-for-human","slug":"llasa-large-multimodal-agent-for-human","title":"LLaSA: A Multimodal LLM for Human Activity Analysis Through Wearable and Smartphone Sensors","date":"2024-06-20","arxiv_id":"2406.14498","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llasa-large-multimodal-agent-for-human#ran","syntology_url":"https://syntology.ai/paper/2406.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14498"}},"official":{"repos":["bashlab/llasa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedical-visual-instruction-tuning-with","slug":"biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","arxiv_id":"2406.13173","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/biomedical-visual-instruction-tuning-with#ran","syntology_url":"https://syntology.ai/paper/2406.13173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13173"}},"official":{"repos":["mao1207/BioMed-VITAL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/finding-blind-spots-in-evaluator-llms-with","slug":"finding-blind-spots-in-evaluator-llms-with","title":"Finding Blind Spots in Evaluator LLMs with Interpretable Checklists","date":"2024-06-19","arxiv_id":"2406.13439","repositories_listed":1,"syntology":null},{"url":"/paper/self-play-with-execution-feedback-improving","slug":"self-play-with-execution-feedback-improving","title":"Self-play with Execution Feedback: Improving Instruction-following Capabilities of Large Language Models","date":"2024-06-19","arxiv_id":"2406.13542","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-play-with-execution-feedback-improving#ran","syntology_url":"https://syntology.ai/paper/2406.13542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13542"}},"official":{"repos":["QwenLM/AutoIF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rs-gpt4v-a-unified-multimodal-instruction","slug":"rs-gpt4v-a-unified-multimodal-instruction","title":"RS-GPT4V: A Unified Multimodal Instruction-Following Dataset for Remote Sensing Image Understanding","date":"2024-06-18","arxiv_id":"2406.12479","repositories_listed":1,"syntology":null},{"url":"/paper/chatbug-a-common-vulnerability-of-aligned","slug":"chatbug-a-common-vulnerability-of-aligned","title":"ChatBug: A Common Vulnerability of Aligned LLMs Induced by Chat Templates","date":"2024-06-17","arxiv_id":"2406.12935","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatbug-a-common-vulnerability-of-aligned#ran","syntology_url":"https://syntology.ai/paper/2406.12935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12935"}},"official":{"repos":["uw-nsl/ChatBug"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-visual-instruction-tuning","slug":"generative-visual-instruction-tuning","title":"Generative Visual Instruction Tuning","date":"2024-06-17","arxiv_id":"2406.11262","repositories_listed":1,"syntology":null},{"url":"/paper/grade-score-quantifying-llm-performance-in","slug":"grade-score-quantifying-llm-performance-in","title":"Grade Score: Quantifying LLM Performance in Option Selection","date":"2024-06-17","arxiv_id":"2406.12043","repositories_listed":1,"syntology":null},{"url":"/paper/refusal-in-language-models-is-mediated-by-a","slug":"refusal-in-language-models-is-mediated-by-a","title":"Refusal in Language Models Is Mediated by a Single Direction","date":"2024-06-17","arxiv_id":"2406.11717","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/refusal-in-language-models-is-mediated-by-a#ran","syntology_url":"https://syntology.ai/paper/2406.11717","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11717"}},"official":{"repos":["andyrdt/refusal_direction"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/wpo-enhancing-rlhf-with-weighted-preference","slug":"wpo-enhancing-rlhf-with-weighted-preference","title":"WPO: Enhancing RLHF with Weighted Preference Optimization","date":"2024-06-17","arxiv_id":"2406.11827","repositories_listed":1,"syntology":null},{"url":"/paper/milora-harnessing-minor-singular-components","slug":"milora-harnessing-minor-singular-components","title":"MiLoRA: Harnessing Minor Singular Components for Parameter-Efficient LLM Finetuning","date":"2024-06-13","arxiv_id":"2406.09044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/milora-harnessing-minor-singular-components#ran","syntology_url":"https://syntology.ai/paper/2406.09044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09044"}},"official":{"repos":["graphpku/pissa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/taste-teaching-large-language-models-to","slug":"taste-teaching-large-language-models-to","title":"TasTe: Teaching Large Language Models to Translate through Self-Reflection","date":"2024-06-12","arxiv_id":"2406.08434","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/taste-teaching-large-language-models-to#ran","syntology_url":"https://syntology.ai/paper/2406.08434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08434"}},"official":{"repos":["yutongwang1216/reflectionllmmt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coevol-constructing-better-responses-for","slug":"coevol-constructing-better-responses-for","title":"CoEvol: Constructing Better Responses for Instruction Finetuning through Multi-Agent Cooperation","date":"2024-06-11","arxiv_id":"2406.07054","repositories_listed":1,"syntology":null},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sciriff-a-resource-to-enhance-language-model","slug":"sciriff-a-resource-to-enhance-language-model","title":"SciRIFF: A Resource to Enhance Language Model Instruction-Following over Scientific Literature","date":"2024-06-10","arxiv_id":"2406.07835","repositories_listed":1,"syntology":null},{"url":"/paper/corda-context-oriented-decomposition","slug":"corda-context-oriented-decomposition","title":"CorDA: Context-Oriented Decomposition Adaptation of Large Language Models for Task-Aware Parameter-Efficient Fine-tuning","date":"2024-06-07","arxiv_id":"2406.05223","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/corda-context-oriented-decomposition#ran","syntology_url":"https://syntology.ai/paper/2406.05223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05223"}},"official":{"repos":["iboing/corda"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/blsp-emo-towards-empathetic-large-speech","slug":"blsp-emo-towards-empathetic-large-speech","title":"BLSP-Emo: Towards Empathetic Large Speech-Language Models","date":"2024-06-06","arxiv_id":"2406.03872","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/blsp-emo-towards-empathetic-large-speech#ran","syntology_url":"https://syntology.ai/paper/2406.03872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03872"}},"official":{"repos":["cwang621/blsp-emo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/genai-arena-an-open-evaluation-platform-for","slug":"genai-arena-an-open-evaluation-platform-for","title":"GenAI Arena: An Open Evaluation Platform for Generative Models","date":"2024-06-06","arxiv_id":"2406.04485","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-moment-matching-distillation-of","slug":"adversarial-moment-matching-distillation-of","title":"Adversarial Moment-Matching Distillation of Large Language Models","date":"2024-06-05","arxiv_id":"2406.02959","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-text-to-image-retrieval-with","slug":"interactive-text-to-image-retrieval-with","title":"Interactive Text-to-Image Retrieval with Large Language Models: A Plug-and-Play Approach","date":"2024-06-05","arxiv_id":"2406.03411","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-text-to-image-retrieval-with#ran","syntology_url":"https://syntology.ai/paper/2406.03411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03411"}},"official":{"repos":["saehyung-lee/plugir"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-as-evaluators-for","slug":"large-language-models-as-evaluators-for","title":"Large Language Models as Evaluators for Recommendation Explanations","date":"2024-06-05","arxiv_id":"2406.03248","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-programming-elicitation-and-repair","slug":"synthetic-programming-elicitation-and-repair","title":"Synthetic Programming Elicitation for Text-to-Code in Very Low-Resource Programming and Formal Languages","date":"2024-06-05","arxiv_id":"2406.03636","repositories_listed":1,"syntology":null},{"url":"/paper/phased-instruction-fine-tuning-for-large","slug":"phased-instruction-fine-tuning-for-large","title":"Phased Instruction Fine-Tuning for Large Language Models","date":"2024-06-01","arxiv_id":"2406.04371","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/phased-instruction-fine-tuning-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.04371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04371"}},"official":{"repos":["xubuvd/phasedsft"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instruction-guided-visual-masking","slug":"instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","arxiv_id":"2405.19783","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":7,"n_ran_checked":11,"n_instrument":9,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"20 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 9 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/instruction-guided-visual-masking#ran","syntology_url":"https://syntology.ai/paper/2405.19783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19783"}},"official":{"repos":["2toinf/ivm"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/is-in-context-learning-sufficient-for","slug":"is-in-context-learning-sufficient-for","title":"Is In-Context Learning Sufficient for Instruction Following in LLMs?","date":"2024-05-30","arxiv_id":"2405.19874","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/is-in-context-learning-sufficient-for#ran","syntology_url":"https://syntology.ai/paper/2405.19874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19874"}},"official":{"repos":["tml-epfl/icl-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/x-instruction-aligning-language-model-in-low","slug":"x-instruction-aligning-language-model-in-low","title":"X-Instruction: Aligning Language Model in Low-resource Languages with Self-curated Cross-lingual Instructions","date":"2024-05-30","arxiv_id":"2405.19744","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/x-instruction-aligning-language-model-in-low#ran","syntology_url":"https://syntology.ai/paper/2405.19744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19744"}},"official":{"repos":["znlp/x-instruction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathchat-benchmarking-mathematical-reasoning","slug":"mathchat-benchmarking-mathematical-reasoning","title":"MathChat: Benchmarking Mathematical Reasoning and Instruction Following in Multi-Turn Interactions","date":"2024-05-29","arxiv_id":"2405.19444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathchat-benchmarking-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2405.19444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19444"}},"official":{"repos":["zhenwen-nlp/mathchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-exploring-language-models-active","slug":"self-exploring-language-models-active","title":"Self-Exploring Language Models: Active Preference Elicitation for Online Alignment","date":"2024-05-29","arxiv_id":"2405.19332","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/self-exploring-language-models-active#ran","syntology_url":"https://syntology.ai/paper/2405.19332","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19332"}},"official":{"repos":["shenao-zhang/selm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/weak-to-strong-search-align-large-language","slug":"weak-to-strong-search-align-large-language","title":"Weak-to-Strong Search: Align Large Language Models via Searching over Small Language Models","date":"2024-05-29","arxiv_id":"2405.19262","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/weak-to-strong-search-align-large-language#ran","syntology_url":"https://syntology.ai/paper/2405.19262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19262"}},"official":{"repos":["zhziszz/weak-to-strong-search"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/empowering-source-free-domain-adaptation-with","slug":"empowering-source-free-domain-adaptation-with","title":"Empowering Source-Free Domain Adaptation with MLLM-driven Curriculum Learning","date":"2024-05-28","arxiv_id":"2405.18376","repositories_listed":1,"syntology":null},{"url":"/paper/promptfix-you-prompt-and-we-fix-the-photo","slug":"promptfix-you-prompt-and-we-fix-the-photo","title":"PromptFix: You Prompt and We Fix the Photo","date":"2024-05-27","arxiv_id":"2405.16785","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":7,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 2 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/promptfix-you-prompt-and-we-fix-the-photo#ran","syntology_url":"https://syntology.ai/paper/2405.16785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16785"}},"official":{"repos":["yeates/promptfix"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-vision-language-action-models-for","slug":"a-survey-on-vision-language-action-models-for","title":"A Survey on Vision-Language-Action Models for Embodied AI","date":"2024-05-23","arxiv_id":"2405.14093","repositories_listed":1,"syntology":null},{"url":"/paper/editworld-simulating-world-dynamics-for","slug":"editworld-simulating-world-dynamics-for","title":"EditWorld: Simulating World Dynamics for Instruction-Following Image Editing","date":"2024-05-23","arxiv_id":"2405.14785","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/editworld-simulating-world-dynamics-for#ran","syntology_url":"https://syntology.ai/paper/2405.14785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14785"}},"official":{"repos":["yangling0818/editworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/disperse-then-merge-pushing-the-limits-of","slug":"disperse-then-merge-pushing-the-limits-of","title":"Disperse-Then-Merge: Pushing the Limits of Instruction Tuning via Alignment Tax Reduction","date":"2024-05-22","arxiv_id":"2405.13432","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/disperse-then-merge-pushing-the-limits-of#ran","syntology_url":"https://syntology.ai/paper/2405.13432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.13432"}},"official":{"repos":["TingchenFu/ACL24-ExpertFusion"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vikhr-the-family-of-open-source-instruction","slug":"vikhr-the-family-of-open-source-instruction","title":"Vikhr: Constructing a State-of-the-art Bilingual Open-Source Instruction-Following Large Language Model for Russian","date":"2024-05-22","arxiv_id":"2405.13929","repositories_listed":1,"syntology":null},{"url":"/paper/recgpt-generative-pre-training-for-text-based","slug":"recgpt-generative-pre-training-for-text-based","title":"RecGPT: Generative Pre-training for Text-based Recommendation","date":"2024-05-21","arxiv_id":"2405.12715","repositories_listed":1,"syntology":null},{"url":"/paper/grounded-3d-llm-with-referent-tokens","slug":"grounded-3d-llm-with-referent-tokens","title":"Grounded 3D-LLM with Referent Tokens","date":"2024-05-16","arxiv_id":"2405.10370","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grounded-3d-llm-with-referent-tokens#ran","syntology_url":"https://syntology.ai/paper/2405.10370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10370"}},"official":{"repos":["OpenRobotLab/Grounded_3D-LLM"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-safety-realignment-framework-via-subspace","slug":"a-safety-realignment-framework-via-subspace","title":"A safety realignment framework via subspace-oriented model fusion for large language models","date":"2024-05-15","arxiv_id":"2405.09055","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-safety-realignment-framework-via-subspace#ran","syntology_url":"https://syntology.ai/paper/2405.09055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09055"}},"official":{"repos":["xinykou/safety_realignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-instruction-following-in-language","slug":"improving-instruction-following-in-language","title":"Improving Instruction Following in Language Models through Proxy-Based Uncertainty Estimation","date":"2024-05-10","arxiv_id":"2405.06424","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-instruction-following-in-language#ran","syntology_url":"https://syntology.ai/paper/2405.06424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06424"}},"official":{"repos":["p-b-u/proxy_based_uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}},"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-llm-guided-counterfactual","slug":"zero-shot-llm-guided-counterfactual","title":"Zero-shot LLM-guided Counterfactual Generation: A Case Study on NLP Model Evaluation","date":"2024-05-08","arxiv_id":"2405.04793","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-llm-guided-counterfactual#ran","syntology_url":"https://syntology.ai/paper/2405.04793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04793"}},"official":{"repos":["AmritaBh/zero-shot-llm-counterfactual"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-as-dataset-analyst-subpopulation","slug":"llm-as-dataset-analyst-subpopulation","title":"LLM as Dataset Analyst: Subpopulation Structure Discovery with Large Language Model","date":"2024-05-03","arxiv_id":"2405.02363","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/llm-as-dataset-analyst-subpopulation#ran","syntology_url":"https://syntology.ai/paper/2405.02363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02363"}},"official":{"repos":["llm-as-dataset-analyst/SSDLLM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/from-complex-to-simple-enhancing-multi","slug":"from-complex-to-simple-enhancing-multi","title":"From Complex to Simple: Enhancing Multi-Constraint Complex Instruction Following Ability of Large Language Models","date":"2024-04-24","arxiv_id":"2404.15846","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/from-complex-to-simple-enhancing-multi#ran","syntology_url":"https://syntology.ai/paper/2404.15846","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15846"}},"official":{"repos":["meowpass/followcomplexinstruction"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/url-universal-referential-knowledge-linking","slug":"url-universal-referential-knowledge-linking","title":"URL: Universal Referential Knowledge Linking via Task-instructed Representation Compression","date":"2024-04-24","arxiv_id":"2404.16248","repositories_listed":1,"syntology":null},{"url":"/paper/meddr-diagnosis-guided-bootstrapping-for","slug":"meddr-diagnosis-guided-bootstrapping-for","title":"GSCo: Towards Generalizable AI in Medicine via Generalist-Specialist Collaboration","date":"2024-04-23","arxiv_id":"2404.15127","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meddr-diagnosis-guided-bootstrapping-for#ran","syntology_url":"https://syntology.ai/paper/2404.15127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15127"}},"official":{"repos":["sunanhe/meddr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-instruction-hierarchy-training-llms-to","slug":"the-instruction-hierarchy-training-llms-to","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","date":"2024-04-19","arxiv_id":"2404.13208","repositories_listed":1,"syntology":null},{"url":"/paper/facial-affective-behavior-analysis-with","slug":"facial-affective-behavior-analysis-with","title":"Facial Affective Behavior Analysis with Instruction Tuning","date":"2024-04-07","arxiv_id":"2404.05052","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/facial-affective-behavior-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2404.05052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05052"}},"official":{"repos":["JackYFL/EmoLA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/teaching-llama-a-new-language-through-cross","slug":"teaching-llama-a-new-language-through-cross","title":"Teaching Llama a New Language Through Cross-Lingual Knowledge Transfer","date":"2024-04-05","arxiv_id":"2404.04042","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/teaching-llama-a-new-language-through-cross#ran","syntology_url":"https://syntology.ai/paper/2404.04042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04042"}},"official":{"repos":["tartunlp/llammas"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-llms-at-detecting-errors-in-llm","slug":"evaluating-llms-at-detecting-errors-in-llm","title":"Evaluating LLMs at Detecting Errors in LLM Responses","date":"2024-04-04","arxiv_id":"2404.03602","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-llms-at-detecting-errors-in-llm#ran","syntology_url":"https://syntology.ai/paper/2404.03602","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03602"}},"official":{"repos":["psunlpgroup/realmistake"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conifer-improving-complex-constrained","slug":"conifer-improving-complex-constrained","title":"Conifer: Improving Complex Constrained Instruction-Following Ability of Large Language Models","date":"2024-04-03","arxiv_id":"2404.02823","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conifer-improving-complex-constrained#ran","syntology_url":"https://syntology.ai/paper/2404.02823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02823"}},"official":{"repos":["coniferlm/conifer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-optimization-of-video-large","slug":"direct-preference-optimization-of-video-large","title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","date":"2024-04-01","arxiv_id":"2404.01258","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-preference-optimization-of-video-large#ran","syntology_url":"https://syntology.ai/paper/2404.01258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01258"}},"official":{"repos":["riflezhang/llava-hound-dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-correction-efficient-tuning-task","slug":"learning-by-correction-efficient-tuning-task","title":"Learning by Correction: Efficient Tuning Task for Zero-Shot Generative Vision-Language Reasoning","date":"2024-04-01","arxiv_id":"2404.00909","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-correction-efficient-tuning-task#ran","syntology_url":"https://syntology.ai/paper/2404.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00909"}},"official":{"repos":["shtuplus/iccc_cvpr2024"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/token-efficient-leverage-learning-in-large","slug":"token-efficient-leverage-learning-in-large","title":"Token-Efficient Leverage Learning in Large Language Models","date":"2024-04-01","arxiv_id":"2404.00914","repositories_listed":1,"syntology":null},{"url":"/paper/coda-constrained-generation-based-data","slug":"coda-constrained-generation-based-data","title":"CoDa: Constrained Generation based Data Augmentation for Low-Resource NLP","date":"2024-03-30","arxiv_id":"2404.00415","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coda-constrained-generation-based-data#ran","syntology_url":"https://syntology.ai/paper/2404.00415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00415"}},"official":{"repos":["sreyan88/coda"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/draw-and-understand-leveraging-visual-prompts","slug":"draw-and-understand-leveraging-visual-prompts","title":"Draw-and-Understand: Leveraging Visual Prompts to Enable MLLMs to Comprehend What You Want","date":"2024-03-29","arxiv_id":"2403.20271","repositories_listed":1,"syntology":null},{"url":"/paper/top-leaderboard-ranking-top-coding","slug":"top-leaderboard-ranking-top-coding","title":"Top Leaderboard Ranking = Top Coding Proficiency, Always? EvoEval: Evolving Coding Benchmarks via LLM","date":"2024-03-28","arxiv_id":"2403.19114","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/top-leaderboard-ranking-top-coding#ran","syntology_url":"https://syntology.ai/paper/2403.19114","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19114"}},"official":{"repos":["evo-eval/evoeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lita-language-instructed-temporal","slug":"lita-language-instructed-temporal","title":"LITA: Language Instructed Temporal-Localization Assistant","date":"2024-03-27","arxiv_id":"2403.19046","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lita-language-instructed-temporal#ran","syntology_url":"https://syntology.ai/paper/2403.19046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19046"}},"official":{"repos":["nvlabs/lita"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flashface-human-image-personalization-with","slug":"flashface-human-image-personalization-with","title":"FlashFace: Human Image Personalization with High-fidelity Identity Preservation","date":"2024-03-25","arxiv_id":"2403.17008","repositories_listed":1,"syntology":{"n":13,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/flashface-human-image-personalization-with#ran","syntology_url":"https://syntology.ai/paper/2403.17008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17008"}},"official":null}},{"url":"/paper/instupr-instruction-based-unsupervised","slug":"instupr-instruction-based-unsupervised","title":"InstUPR : Instruction-based Unsupervised Passage Reranking with Large Language Models","date":"2024-03-25","arxiv_id":"2403.16435","repositories_listed":1,"syntology":null},{"url":"/paper/rl-for-consistency-models-faster-reward","slug":"rl-for-consistency-models-faster-reward","title":"RL for Consistency Models: Faster Reward Guided Text-to-Image Generation","date":"2024-03-25","arxiv_id":"2404.03673","repositories_listed":1,"syntology":null},{"url":"/paper/wangchanlion-and-wangchanx-mrc-eval","slug":"wangchanlion-and-wangchanx-mrc-eval","title":"WangchanLion and WangchanX MRC Eval","date":"2024-03-24","arxiv_id":"2403.16127","repositories_listed":1,"syntology":null},{"url":"/paper/building-accurate-translation-tailored-llms","slug":"building-accurate-translation-tailored-llms","title":"Building Accurate Translation-Tailored LLMs with Language Aware Instruction Tuning","date":"2024-03-21","arxiv_id":"2403.14399","repositories_listed":1,"syntology":null},{"url":"/paper/mmidr-teaching-large-language-model-to","slug":"mmidr-teaching-large-language-model-to","title":"MMIDR: Teaching Large Language Model to Interpret Multimodal Misinformation via Knowledge Distillation","date":"2024-03-21","arxiv_id":"2403.14171","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-spot-interactive-reasoning-improves","slug":"chain-of-spot-interactive-reasoning-improves","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","date":"2024-03-19","arxiv_id":"2403.12966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chain-of-spot-interactive-reasoning-improves#ran","syntology_url":"https://syntology.ai/paper/2403.12966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12966"}},"official":{"repos":["dongyh20/chain-of-spot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/third-party-language-model-performance","slug":"third-party-language-model-performance","title":"Third-Party Language Model Performance Prediction from Instruction","date":"2024-03-19","arxiv_id":"2403.12413","repositories_listed":1,"syntology":null},{"url":"/paper/minedreamer-learning-to-follow-instructions","slug":"minedreamer-learning-to-follow-instructions","title":"MineDreamer: Learning to Follow Instructions via Chain-of-Imagination for Simulated-World Control","date":"2024-03-18","arxiv_id":"2403.12037","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/minedreamer-learning-to-follow-instructions#ran","syntology_url":"https://syntology.ai/paper/2403.12037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12037"}},"official":{"repos":["Zhoues/MineDreamer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chartinstruct-instruction-tuning-for-chart","slug":"chartinstruct-instruction-tuning-for-chart","title":"ChartInstruct: Instruction Tuning for Chart Comprehension and Reasoning","date":"2024-03-14","arxiv_id":"2403.09028","repositories_listed":1,"syntology":null},{"url":"/paper/coin-a-benchmark-of-continual-instruction","slug":"coin-a-benchmark-of-continual-instruction","title":"CoIN: A Benchmark of Continual Instruction tuNing for Multimodel Large Language Model","date":"2024-03-13","arxiv_id":"2403.08350","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coin-a-benchmark-of-continual-instruction#ran","syntology_url":"https://syntology.ai/paper/2403.08350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08350"}},"official":{"repos":["zackschen/coin"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"adc0245c0d097e2809700d0967148b05fa47a273bab2b2c27c09ef9dc5b7686b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}