{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/instruction-following/papers/3","list_of":"/task/instruction-following","task":"Instruction Following","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":12,"rows_per_page":100,"rows":[201,300],"of":1135,"counts":{"archive_papers_tagged":1135,"with_a_code_link":609,"where_syntology_ran_a_sample":311,"not_listed_spam_title":0,"listed":1135,"listed_where_code_ran":311,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":255,"every_run_a_failure_of_syntologys_instrument":56,"listed_with_a_run_with_no_instrument_failure":255,"listed_every_run_a_failure_of_syntologys_instrument":56,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/instruction-following","prev":"/task/instruction-following/papers/2","next":"/task/instruction-following/papers/4","papers":[{"url":"/paper/internlm-xcomposer2-5-reward-a-simple-yet","slug":"internlm-xcomposer2-5-reward-a-simple-yet","title":"InternLM-XComposer2.5-Reward: A Simple Yet Effective Multi-Modal Reward Model","date":"2025-01-21","arxiv_id":"2501.12368","repositories_listed":1,"syntology":null},{"url":"/paper/curiosity-driven-reinforcement-learning-from","slug":"curiosity-driven-reinforcement-learning-from","title":"Curiosity-Driven Reinforcement Learning from Human Feedback","date":"2025-01-20","arxiv_id":"2501.11463","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/curiosity-driven-reinforcement-learning-from#ran","syntology_url":"https://syntology.ai/paper/2501.11463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.11463"}},"official":{"repos":["ernie-research/cd-rlhf"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-multi-modal-ai-copilot-for-single-cell","slug":"a-multi-modal-ai-copilot-for-single-cell","title":"A Multi-Modal AI Copilot for Single-Cell Analysis with Instruction Following","date":"2025-01-14","arxiv_id":"2501.08187","repositories_listed":1,"syntology":null},{"url":"/paper/facial-dynamics-in-video-instruction-tuning","slug":"facial-dynamics-in-video-instruction-tuning","title":"Facial Dynamics in Video: Instruction Tuning for Improved Facial Expression Perception and Contextual Awareness","date":"2025-01-14","arxiv_id":"2501.07978","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-label-refinement-matters-more-than","slug":"iterative-label-refinement-matters-more-than","title":"Iterative Label Refinement Matters More than Preference Optimization under Weak Supervision","date":"2025-01-14","arxiv_id":"2501.07886","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-domain-adaptive-post-training","slug":"demystifying-domain-adaptive-post-training","title":"Demystifying Domain-adaptive Post-training for Financial LLMs","date":"2025-01-09","arxiv_id":"2501.04961","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/demystifying-domain-adaptive-post-training#ran","syntology_url":"https://syntology.ai/paper/2501.04961","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04961"}},"official":{"repos":["salesforceairesearch/findap"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/progco-program-helps-self-correction-of-large","slug":"progco-program-helps-self-correction-of-large","title":"ProgCo: Program Helps Self-Correction of Large Language Models","date":"2025-01-02","arxiv_id":"2501.01264","repositories_listed":1,"syntology":null},{"url":"/paper/towards-interactive-deepfake-analysis","slug":"towards-interactive-deepfake-analysis","title":"Towards Interactive Deepfake Analysis","date":"2025-01-02","arxiv_id":"2501.01164","repositories_listed":1,"syntology":null},{"url":"/paper/tinyhelen-s-first-curriculum-training-and","slug":"tinyhelen-s-first-curriculum-training-and","title":"TinyHelen's First Curriculum: Training and Evaluating Tiny Language Models in a Simpler Language Environment","date":"2024-12-31","arxiv_id":"2501.00522","repositories_listed":1,"syntology":null},{"url":"/paper/find-the-intention-of-instruction","slug":"find-the-intention-of-instruction","title":"Find the Intention of Instruction: Comprehensive Evaluation of Instruction Understanding for Large Language Models","date":"2024-12-27","arxiv_id":"2412.19450","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/find-the-intention-of-instruction#ran","syntology_url":"https://syntology.ai/paper/2412.19450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.19450"}},"official":{"repos":["hyeonseokk/ioinst"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/align-anything-training-all-modality-models","slug":"align-anything-training-all-modality-models","title":"Align Anything: Training All-Modality Models to Follow Instructions with Language Feedback","date":"2024-12-20","arxiv_id":"2412.15838","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/align-anything-training-all-modality-models#ran","syntology_url":"https://syntology.ai/paper/2412.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15838"}},"official":{"repos":["pku-alignment/align-anything"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/href-human-response-guided-evaluation-of","slug":"href-human-response-guided-evaluation-of","title":"HREF: Human Response-Guided Evaluation of Instruction Following in Language Models","date":"2024-12-20","arxiv_id":"2412.15524","repositories_listed":1,"syntology":null},{"url":"/paper/llm-rg4-flexible-and-factual-radiology-report","slug":"llm-rg4-flexible-and-factual-radiology-report","title":"LLM-RG4: Flexible and Factual Radiology Report Generation across Diverse Input Contexts","date":"2024-12-16","arxiv_id":"2412.12001","repositories_listed":1,"syntology":null},{"url":"/paper/spar-self-play-with-tree-search-refinement-to","slug":"spar-self-play-with-tree-search-refinement-to","title":"SPaR: Self-Play with Tree-Search Refinement to Improve Instruction-Following in Large Language Models","date":"2024-12-16","arxiv_id":"2412.11605","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/spar-self-play-with-tree-search-refinement-to#ran","syntology_url":"https://syntology.ai/paper/2412.11605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11605"}},"official":{"repos":["thu-coai/spar"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-instruction-tuning-with-500x-fewer","slug":"visual-instruction-tuning-with-500x-fewer","title":"LLaVA Steering: Visual Instruction Tuning with 500x Fewer Parameters through Modality Linear Representation-Steering","date":"2024-12-16","arxiv_id":"2412.12359","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/visual-instruction-tuning-with-500x-fewer#ran","syntology_url":"https://syntology.ai/paper/2412.12359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12359"}},"official":{"repos":["bibisbar/LLaVA-Steering"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pediabench-a-comprehensive-chinese-pediatric","slug":"pediabench-a-comprehensive-chinese-pediatric","title":"PediaBench: A Comprehensive Chinese Pediatric Dataset for Benchmarking Large Language Models","date":"2024-12-09","arxiv_id":"2412.06287","repositories_listed":1,"syntology":null},{"url":"/paper/sloth-scaling-laws-for-llm-skills-to-predict","slug":"sloth-scaling-laws-for-llm-skills-to-predict","title":"Sloth: scaling laws for LLM skills to predict multi-benchmark performance across families","date":"2024-12-09","arxiv_id":"2412.06540","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sloth-scaling-laws-for-llm-skills-to-predict#ran","syntology_url":"https://syntology.ai/paper/2412.06540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06540"}},"official":{"repos":["felipemaiapolo/sloth"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kasa-knowledge-aware-singular-value","slug":"kasa-knowledge-aware-singular-value","title":"KaSA: Knowledge-Aware Singular-Value Adaptation of Large Language Models","date":"2024-12-08","arxiv_id":"2412.06071","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/kasa-knowledge-aware-singular-value#ran","syntology_url":"https://syntology.ai/paper/2412.06071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06071"}},"official":{"repos":["juyongjiang/kasa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/compositional-image-retrieval-via-instruction","slug":"compositional-image-retrieval-via-instruction","title":"Compositional Image Retrieval via Instruction-Aware Contrastive Learning","date":"2024-12-07","arxiv_id":"2412.05756","repositories_listed":1,"syntology":null},{"url":"/paper/rsunivlm-a-unified-vision-language-model-for","slug":"rsunivlm-a-unified-vision-language-model-for","title":"RSUniVLM: A Unified Vision Language Model for Remote Sensing via Granularity-oriented Mixture of Experts","date":"2024-12-07","arxiv_id":"2412.05679","repositories_listed":1,"syntology":null},{"url":"/paper/prefixkv-adaptive-prefix-kv-cache-is-what","slug":"prefixkv-adaptive-prefix-kv-cache-is-what","title":"PrefixKV: Adaptive Prefix KV Cache is What Vision Instruction-Following Models Need for Efficient Generation","date":"2024-12-04","arxiv_id":"2412.03409","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prefixkv-adaptive-prefix-kv-cache-is-what#ran","syntology_url":"https://syntology.ai/paper/2412.03409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03409"}},"official":{"repos":["THU-MIG/PrefixKV"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/agri-llava-knowledge-infused-large-multimodal","slug":"agri-llava-knowledge-infused-large-multimodal","title":"Agri-LLaVA: Knowledge-Infused Large Multimodal Assistant on Agricultural Pests and Diseases","date":"2024-12-03","arxiv_id":"2412.02158","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agri-llava-knowledge-infused-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2412.02158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02158"}},"official":{"repos":["kki2eve/agri-llava"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/showui-one-vision-language-action-model-for","slug":"showui-one-vision-language-action-model-for","title":"ShowUI: One Vision-Language-Action Model for GUI Visual Agent","date":"2024-11-26","arxiv_id":"2411.17465","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/showui-one-vision-language-action-model-for#ran","syntology_url":"https://syntology.ai/paper/2411.17465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17465"}},"official":{"repos":["showlab/showui"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-instruction-tuning-an","slug":"parameter-efficient-instruction-tuning-an","title":"Parameter Efficient Instruction Tuning: An Empirical Study","date":"2024-11-25","arxiv_id":"2411.16775","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parameter-efficient-instruction-tuning-an#ran","syntology_url":"https://syntology.ai/paper/2411.16775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16775"}},"official":{"repos":["AdaBit-AI/parameter_efficient_instruction_tuning"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/geoground-a-unified-large-vision-language","slug":"geoground-a-unified-large-vision-language","title":"GeoGround: A Unified Large Vision-Language Model for Remote Sensing Visual Grounding","date":"2024-11-16","arxiv_id":"2411.11904","repositories_listed":1,"syntology":null},{"url":"/paper/mpoxvlm-a-vision-language-model-for","slug":"mpoxvlm-a-vision-language-model-for","title":"MpoxVLM: A Vision-Language Model for Diagnosing Skin Lesions from Mpox Virus Infection","date":"2024-11-16","arxiv_id":"2411.10888","repositories_listed":1,"syntology":null},{"url":"/paper/mlan-language-based-instruction-tuning","slug":"mlan-language-based-instruction-tuning","title":"MLAN: Language-Based Instruction Tuning Improves Zero-Shot Generalization of Multimodal Large Language Models","date":"2024-11-15","arxiv_id":"2411.10557","repositories_listed":1,"syntology":null},{"url":"/paper/lhrs-bot-nova-improved-multimodal-large","slug":"lhrs-bot-nova-improved-multimodal-large","title":"LHRS-Bot-Nova: Improved Multimodal Large Language Model for Remote Sensing Vision-Language Interpretation","date":"2024-11-14","arxiv_id":"2411.09301","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lhrs-bot-nova-improved-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2411.09301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.09301"}},"official":{"repos":["NJU-LHRS/LHRS-Bot"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lifbench-evaluating-the-instruction-following","slug":"lifbench-evaluating-the-instruction-following","title":"LIFBench: Evaluating the Instruction Following Performance and Stability of Large Language Models in Long-Context Scenarios","date":"2024-11-11","arxiv_id":"2411.07037","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lifbench-evaluating-the-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2411.07037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07037"}},"official":{"repos":["sheldonwu0327/lif-bench-2024"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/setlexsem-challenge-using-set-operations-to","slug":"setlexsem-challenge-using-set-operations-to","title":"SetLexSem Challenge: Using Set Operations to Evaluate the Lexical and Semantic Robustness of Language Models","date":"2024-11-11","arxiv_id":"2411.07336","repositories_listed":1,"syntology":{"n":23,"n_ran":19,"n_constructed":0,"n_ran_checked":16,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/setlexsem-challenge-using-set-operations-to#ran","syntology_url":"https://syntology.ai/paper/2411.07336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07336"}},"official":{"repos":["amazon-science/setlexsem-challenge"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/iopo-empowering-llms-with-complex-instruction","slug":"iopo-empowering-llms-with-complex-instruction","title":"IOPO: Empowering LLMs with Complex Instruction Following via Input-Output Preference Optimization","date":"2024-11-09","arxiv_id":"2411.06208","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-calibration-of-win-rate-estimation","slug":"bayesian-calibration-of-win-rate-estimation","title":"Bayesian Calibration of Win Rate Estimation with LLM Evaluators","date":"2024-11-07","arxiv_id":"2411.04424","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bayesian-calibration-of-win-rate-estimation#ran","syntology_url":"https://syntology.ai/paper/2411.04424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04424"}},"official":{"repos":["yale-nlp/bay-calibration-llm-evaluators"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-loss-of-context-awareness-in-general","slug":"on-the-loss-of-context-awareness-in-general","title":"On the Loss of Context-awareness in General Instruction Fine-tuning","date":"2024-11-05","arxiv_id":"2411.02688","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-loss-of-context-awareness-in-general#ran","syntology_url":"https://syntology.ai/paper/2411.02688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02688"}},"official":{"repos":["YihanWang617/context_awareness"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/rate-explain-and-cite-rec-enhanced","slug":"rate-explain-and-cite-rec-enhanced","title":"Rate, Explain and Cite (REC): Enhanced Explanation and Attribution in Automatic Evaluation by Large Language Models","date":"2024-11-03","arxiv_id":"2411.02448","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-content-relevance-evaluating","slug":"beyond-content-relevance-evaluating","title":"Beyond Content Relevance: Evaluating Instruction Following in Retrieval Models","date":"2024-10-31","arxiv_id":"2410.23841","repositories_listed":1,"syntology":null},{"url":"/paper/constraint-back-translation-improves-complex","slug":"constraint-back-translation-improves-complex","title":"Constraint Back-translation Improves Complex Instruction Following of Large Language Models","date":"2024-10-31","arxiv_id":"2410.24175","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/constraint-back-translation-improves-complex#ran","syntology_url":"https://syntology.ai/paper/2410.24175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24175"}},"official":{"repos":["thu-keg/crab"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llamo-large-language-model-based-molecular","slug":"llamo-large-language-model-based-molecular","title":"LLaMo: Large Language Model-based Molecular Graph Assistant","date":"2024-10-31","arxiv_id":"2411.00871","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llamo-large-language-model-based-molecular#ran","syntology_url":"https://syntology.ai/paper/2411.00871","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00871"}},"official":{"repos":["mlvlab/llamo"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mdcure-a-scalable-pipeline-for-multi-document","slug":"mdcure-a-scalable-pipeline-for-multi-document","title":"MDCure: A Scalable Pipeline for Multi-Document Instruction-Following","date":"2024-10-30","arxiv_id":"2410.23463","repositories_listed":1,"syntology":null},{"url":"/paper/falcon-feedback-driven-adaptive-long-short","slug":"falcon-feedback-driven-adaptive-long-short","title":"FALCON: Feedback-driven Adaptive Long/short-term memory reinforced Coding Optimization system","date":"2024-10-28","arxiv_id":"2410.21349","repositories_listed":1,"syntology":null},{"url":"/paper/decore-decoding-by-contrasting-retrieval","slug":"decore-decoding-by-contrasting-retrieval","title":"DeCoRe: Decoding by Contrasting Retrieval Heads to Mitigate Hallucinations","date":"2024-10-24","arxiv_id":"2410.18860","repositories_listed":1,"syntology":null},{"url":"/paper/open6dor-benchmarking-open-instruction-6-dof","slug":"open6dor-benchmarking-open-instruction-6-dof","title":"Open6DOR: Benchmarking Open-instruction 6-DoF Object Rearrangement and A VLM-based Approach","date":"2024-10-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adem-vl-adaptive-and-embedded-fusion-for","slug":"adem-vl-adaptive-and-embedded-fusion-for","title":"ADEM-VL: Adaptive and Embedded Fusion for Efficient Vision-Language Tuning","date":"2024-10-23","arxiv_id":"2410.17779","repositories_listed":1,"syntology":null},{"url":"/paper/asynchronous-rlhf-faster-and-more-efficient","slug":"asynchronous-rlhf-faster-and-more-efficient","title":"Asynchronous RLHF: Faster and More Efficient Off-Policy RL for Language Models","date":"2024-10-23","arxiv_id":"2410.18252","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/asynchronous-rlhf-faster-and-more-efficient#ran","syntology_url":"https://syntology.ai/paper/2410.18252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18252"}},"official":{"repos":["mnoukhov/async_rlhf"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/cross-lingual-transfer-of-reward-models-in","slug":"cross-lingual-transfer-of-reward-models-in","title":"Cross-lingual Transfer of Reward Models in Multilingual Alignment","date":"2024-10-23","arxiv_id":"2410.18027","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-lingual-transfer-of-reward-models-in#ran","syntology_url":"https://syntology.ai/paper/2410.18027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18027"}},"official":{"repos":["iq-kaist/rm-lingual-transfer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-model-control-improving-multiple-large","slug":"cross-model-control-improving-multiple-large","title":"Cross-model Control: Improving Multiple Large Language Models in One-time Training","date":"2024-10-23","arxiv_id":"2410.17599","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cross-model-control-improving-multiple-large#ran","syntology_url":"https://syntology.ai/paper/2410.17599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17599"}},"official":{"repos":["wujwyi/cmc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/griffon-g-bridging-vision-language-and-vision","slug":"griffon-g-bridging-vision-language-and-vision","title":"Griffon-G: Bridging Vision-Language and Vision-Centric Tasks via Large Multimodal Models","date":"2024-10-21","arxiv_id":"2410.16163","repositories_listed":1,"syntology":null},{"url":"/paper/multi-if-benchmarking-llms-on-multi-turn-and","slug":"multi-if-benchmarking-llms-on-multi-turn-and","title":"Multi-IF: Benchmarking LLMs on Multi-Turn and Multilingual Instructions Following","date":"2024-10-21","arxiv_id":"2410.15553","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-if-benchmarking-llms-on-multi-turn-and#ran","syntology_url":"https://syntology.ai/paper/2410.15553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15553"}},"official":{"repos":["facebookresearch/Multi-IF"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selecting-influential-samples-for-long","slug":"selecting-influential-samples-for-long","title":"GATEAU: Selecting Influential Samples for Long Context Alignment","date":"2024-10-21","arxiv_id":"2410.15633","repositories_listed":1,"syntology":null},{"url":"/paper/do-llms-estimate-uncertainty-well-in","slug":"do-llms-estimate-uncertainty-well-in","title":"Do LLMs estimate uncertainty well in instruction-following?","date":"2024-10-18","arxiv_id":"2410.14582","repositories_listed":1,"syntology":null},{"url":"/paper/do-llms-know-internally-when-they-follow","slug":"do-llms-know-internally-when-they-follow","title":"Do LLMs \"know\" internally when they follow instructions?","date":"2024-10-18","arxiv_id":"2410.14516","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/do-llms-know-internally-when-they-follow#ran","syntology_url":"https://syntology.ai/paper/2410.14516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14516"}},"official":{"repos":["apple/ml-internal-llms-instruction-following"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/logu-long-form-generation-with-uncertainty","slug":"logu-long-form-generation-with-uncertainty","title":"LoGU: Long-form Generation with Uncertainty Expressions","date":"2024-10-18","arxiv_id":"2410.14309","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/logu-long-form-generation-with-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2410.14309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14309"}},"official":{"repos":["rhyang2021/logu"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/loldu-low-rank-adaptation-via-lower-diag","slug":"loldu-low-rank-adaptation-via-lower-diag","title":"LoLDU: Low-Rank Adaptation via Lower-Diag-Upper Decomposition for Parameter-Efficient Fine-Tuning","date":"2024-10-17","arxiv_id":"2410.13618","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/loldu-low-rank-adaptation-via-lower-diag#ran","syntology_url":"https://syntology.ai/paper/2410.13618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13618"}},"official":{"repos":["skddj/loldu"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-instruction-following","slug":"evaluating-the-instruction-following","title":"Evaluating the Instruction-following Abilities of Language Models using Knowledge Tasks","date":"2024-10-16","arxiv_id":"2410.12972","repositories_listed":1,"syntology":null},{"url":"/paper/meta-chunking-learning-efficient-text","slug":"meta-chunking-learning-efficient-text","title":"Meta-Chunking: Learning Text Segmentation and Semantic Completion via Logical Perception","date":"2024-10-16","arxiv_id":"2410.12788","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/meta-chunking-learning-efficient-text#ran","syntology_url":"https://syntology.ai/paper/2410.12788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12788"}},"official":{"repos":["IAAR-Shanghai/Meta-Chunking"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rulerag-rule-guided-retrieval-augmented","slug":"rulerag-rule-guided-retrieval-augmented","title":"RuleRAG: Rule-guided retrieval-augmented generation with language models for question answering","date":"2024-10-15","arxiv_id":"2410.22353","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-leverage-demonstration-data-in","slug":"how-to-leverage-demonstration-data-in","title":"How to Leverage Demonstration Data in Alignment for Large Language Model? A Self-Imitation Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10093","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-leverage-demonstration-data-in#ran","syntology_url":"https://syntology.ai/paper/2410.10093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10093"}},"official":{"repos":["tengxiao1/gsil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-general-instruction-following","slug":"toward-general-instruction-following","title":"Toward General Instruction-Following Alignment for Retrieval-Augmented Generation","date":"2024-10-12","arxiv_id":"2410.09584","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/toward-general-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2410.09584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09584"}},"official":{"repos":["dongguanting/FollowRAG"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-imbalance-driven-rewarding-for","slug":"language-imbalance-driven-rewarding-for","title":"Language Imbalance Driven Rewarding for Multilingual Self-improving","date":"2024-10-11","arxiv_id":"2410.08964","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/language-imbalance-driven-rewarding-for#ran","syntology_url":"https://syntology.ai/paper/2410.08964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08964"}},"official":{"repos":["znlp/language-imbalance-driven-rewarding"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/copesd-a-multi-level-surgical-motion-dataset","slug":"copesd-a-multi-level-surgical-motion-dataset","title":"CoPESD: A Multi-Level Surgical Motion Dataset for Training Large Vision-Language Models to Co-Pilot Endoscopic Submucosal Dissection","date":"2024-10-10","arxiv_id":"2410.07540","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/copesd-a-multi-level-surgical-motion-dataset#ran","syntology_url":"https://syntology.ai/paper/2410.07540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07540"}},"official":{"repos":["gkw0010/copesd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-augmented-data-enhances-direct","slug":"reward-augmented-data-enhances-direct","title":"Reward-Augmented Data Enhances Direct Preference Alignment of LLMs","date":"2024-10-10","arxiv_id":"2410.08067","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reward-augmented-data-enhances-direct#ran","syntology_url":"https://syntology.ai/paper/2410.08067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08067"}},"official":{"repos":["shenao-zhang/reward-augmented-preference"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reife-re-evaluating-instruction-following","slug":"reife-re-evaluating-instruction-following","title":"ReIFE: Re-evaluating Instruction-Following Evaluation","date":"2024-10-09","arxiv_id":"2410.07069","repositories_listed":1,"syntology":null},{"url":"/paper/aria-an-open-multimodal-native-mixture-of","slug":"aria-an-open-multimodal-native-mixture-of","title":"Aria: An Open Multimodal Native Mixture-of-Experts Model","date":"2024-10-08","arxiv_id":"2410.05993","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aria-an-open-multimodal-native-mixture-of#ran","syntology_url":"https://syntology.ai/paper/2410.05993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05993"}},"official":{"repos":["rhymes-ai/aria"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/teochat-a-large-vision-language-assistant-for","slug":"teochat-a-large-vision-language-assistant-for","title":"TEOChat: A Large Vision-Language Assistant for Temporal Earth Observation Data","date":"2024-10-08","arxiv_id":"2410.06234","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":5,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/teochat-a-large-vision-language-assistant-for#ran","syntology_url":"https://syntology.ai/paper/2410.06234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06234"}},"official":{"repos":["ermongroup/teochat"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/a-recipe-for-building-a-compliant-real-estate","slug":"a-recipe-for-building-a-compliant-real-estate","title":"A Recipe For Building a Compliant Real Estate Chatbot","date":"2024-10-07","arxiv_id":"2410.10860","repositories_listed":1,"syntology":null},{"url":"/paper/cs4-measuring-the-creativity-of-large","slug":"cs4-measuring-the-creativity-of-large","title":"CS4: Measuring the Creativity of Large Language Models Automatically by Controlling the Number of Story-Writing Constraints","date":"2024-10-05","arxiv_id":"2410.04197","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cs4-measuring-the-creativity-of-large#ran","syntology_url":"https://syntology.ai/paper/2410.04197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04197"}},"official":{"repos":["anirudhlakkaraju/cs4_benchmark"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/commonit-commonality-aware-instruction-tuning","slug":"commonit-commonality-aware-instruction-tuning","title":"CommonIT: Commonality-Aware Instruction Tuning for Large Language Models via Data Partitions","date":"2024-10-04","arxiv_id":"2410.03077","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/commonit-commonality-aware-instruction-tuning#ran","syntology_url":"https://syntology.ai/paper/2410.03077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03077"}},"official":{"repos":["raojay7/commonit"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-powered-llm-modality-expansion-for-large","slug":"self-powered-llm-modality-expansion-for-large","title":"Self-Powered LLM Modality Expansion for Large Speech-Text Models","date":"2024-10-04","arxiv_id":"2410.03798","repositories_listed":1,"syntology":null},{"url":"/paper/laser-learning-to-adaptively-select-reward","slug":"laser-learning-to-adaptively-select-reward","title":"LASeR: Learning to Adaptively Select Reward Models with Multi-Armed Bandits","date":"2024-10-02","arxiv_id":"2410.01735","repositories_listed":1,"syntology":null},{"url":"/paper/medqa-cs-benchmarking-large-language-models","slug":"medqa-cs-benchmarking-large-language-models","title":"MedQA-CS: Benchmarking Large Language Models Clinical Skills Using an AI-SCE Framework","date":"2024-10-02","arxiv_id":"2410.01553","repositories_listed":1,"syntology":null},{"url":"/paper/developing-instruction-following-speech","slug":"developing-instruction-following-speech","title":"DeSTA2: Developing Instruction-Following Speech Language Model Without Speech Instruction-Tuning Data","date":"2024-09-30","arxiv_id":"2409.20007","repositories_listed":1,"syntology":null},{"url":"/paper/robin3d-improving-3d-large-language-model-via","slug":"robin3d-improving-3d-large-language-model-via","title":"Robin3D: Improving 3D Large Language Model via Robust Instruction Tuning","date":"2024-09-30","arxiv_id":"2410.00255","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robin3d-improving-3d-large-language-model-via#ran","syntology_url":"https://syntology.ai/paper/2410.00255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00255"}},"official":{"repos":["weitaikang/robin3d"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/align-2-llava-cascaded-human-and-large","slug":"align-2-llava-cascaded-human-and-large","title":"Align$^2$LLaVA: Cascaded Human and Large Language Model Preference Alignment for Multi-modal Instruction Curation","date":"2024-09-27","arxiv_id":"2409.18541","repositories_listed":1,"syntology":null},{"url":"/paper/ruler-a-model-agnostic-method-to-control","slug":"ruler-a-model-agnostic-method-to-control","title":"Ruler: A Model-Agnostic Method to Control Generated Length for Large Language Models","date":"2024-09-27","arxiv_id":"2409.18943","repositories_listed":1,"syntology":null},{"url":"/paper/infer-human-s-intentions-before-following","slug":"infer-human-s-intentions-before-following","title":"Infer Human's Intentions Before Following Natural Language Instructions","date":"2024-09-26","arxiv_id":"2409.18073","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/infer-human-s-intentions-before-following#ran","syntology_url":"https://syntology.ai/paper/2409.18073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18073"}},"official":{"repos":["simon-wan/fiser"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/eventhallusion-diagnosing-event","slug":"eventhallusion-diagnosing-event","title":"EventHallusion: Diagnosing Event Hallucinations in Video LLMs","date":"2024-09-25","arxiv_id":"2409.16597","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eventhallusion-diagnosing-event#ran","syntology_url":"https://syntology.ai/paper/2409.16597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16597"}},"official":{"repos":["stevetich/eventhallusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mitigating-the-bias-of-large-language-model","slug":"mitigating-the-bias-of-large-language-model","title":"Mitigating the Bias of Large Language Model Evaluation","date":"2024-09-25","arxiv_id":"2409.16788","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-the-bias-of-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2409.16788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16788"}},"official":{"repos":["Joe-Hall-Lee/Debias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fmdllama-financial-misinformation-detection","slug":"fmdllama-financial-misinformation-detection","title":"FMDLlama: Financial Misinformation Detection based on Large Language Models","date":"2024-09-24","arxiv_id":"2409.16452","repositories_listed":1,"syntology":null},{"url":"/paper/mm-camobj-a-comprehensive-multimodal-dataset","slug":"mm-camobj-a-comprehensive-multimodal-dataset","title":"MM-CamObj: A Comprehensive Multimodal Dataset for Camouflaged Object Scenarios","date":"2024-09-24","arxiv_id":"2409.16084","repositories_listed":1,"syntology":null},{"url":"/paper/archon-an-architecture-search-framework-for","slug":"archon-an-architecture-search-framework-for","title":"Archon: An Architecture Search Framework for Inference-Time Techniques","date":"2024-09-23","arxiv_id":"2409.15254","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/archon-an-architecture-search-framework-for#ran","syntology_url":"https://syntology.ai/paper/2409.15254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15254"}},"official":{"repos":["scalingintelligence/archon"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/omnibench-towards-the-future-of-universal","slug":"omnibench-towards-the-future-of-universal","title":"OmniBench: Towards The Future of Universal Omni-Language Models","date":"2024-09-23","arxiv_id":"2409.15272","repositories_listed":1,"syntology":null},{"url":"/paper/style-over-substance-failure-modes-of-llm","slug":"style-over-substance-failure-modes-of-llm","title":"Style Outweighs Substance: Failure Modes of LLM Judges in Alignment Benchmarking","date":"2024-09-23","arxiv_id":"2409.15268","repositories_listed":1,"syntology":null},{"url":"/paper/toolplanner-a-tool-augmented-llm-for-multi","slug":"toolplanner-a-tool-augmented-llm-for-multi","title":"ToolPlanner: A Tool Augmented LLM for Multi Granularity Instructions with Path Planning and Feedback","date":"2024-09-23","arxiv_id":"2409.14826","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/toolplanner-a-tool-augmented-llm-for-multi#ran","syntology_url":"https://syntology.ai/paper/2409.14826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14826"}},"official":{"repos":["xiaomi/toolplanner"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2409-13989","slug":"2409-13989","title":"ChemEval: A Comprehensive Multi-Level Chemical Evaluation for Large Language Models","date":"2024-09-21","arxiv_id":"2409.13989","repositories_listed":1,"syntology":{"n":18,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":18,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/2409-13989#ran","syntology_url":"https://syntology.ai/paper/2409.13989","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.13989"}},"official":{"repos":["ustc-starteam/chemeval"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2409-14247","slug":"2409-14247","title":"Repairs in a Block World: A New Benchmark for Handling User Corrections with Multi-Modal Language Models","date":"2024-09-21","arxiv_id":"2409.14247","repositories_listed":1,"syntology":null},{"url":"/paper/2409-14254","slug":"2409-14254","title":"Instruction Following without Instruction Tuning","date":"2024-09-21","arxiv_id":"2409.14254","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-quantized","slug":"a-comprehensive-evaluation-of-quantized","title":"Exploring the Trade-Offs: Quantization Methods, Task Difficulty, and Model Size in Large Language Models From Edge to Giant","date":"2024-09-17","arxiv_id":"2409.11055","repositories_listed":1,"syntology":null},{"url":"/paper/diversify-and-conquer-diversity-centric-data","slug":"diversify-and-conquer-diversity-centric-data","title":"Diversify and Conquer: Diversity-Centric Data Selection with Iterative Refinement","date":"2024-09-17","arxiv_id":"2409.11378","repositories_listed":1,"syntology":null},{"url":"/paper/asft-aligned-supervised-fine-tuning-through","slug":"asft-aligned-supervised-fine-tuning-through","title":"ASFT: Aligned Supervised Fine-Tuning through Absolute Likelihood","date":"2024-09-14","arxiv_id":"2409.10571","repositories_listed":1,"syntology":null},{"url":"/paper/adappa-adaptive-position-pre-fill-jailbreak","slug":"adappa-adaptive-position-pre-fill-jailbreak","title":"AdaPPA: Adaptive Position Pre-Fill Jailbreak Attack Approach Targeting LLMs","date":"2024-09-11","arxiv_id":"2409.07503","repositories_listed":1,"syntology":null},{"url":"/paper/genagent-build-collaborative-ai-systems-with","slug":"genagent-build-collaborative-ai-systems-with","title":"ComfyBench: Benchmarking LLM-based Agents in ComfyUI for Autonomously Designing Collaborative AI Systems","date":"2024-09-02","arxiv_id":"2409.01392","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genagent-build-collaborative-ai-systems-with#ran","syntology_url":"https://syntology.ai/paper/2409.01392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01392"}},"official":{"repos":["xxyQwQ/ComfyBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-judge-selective-instruction-following","slug":"self-judge-selective-instruction-following","title":"Self-Judge: Selective Instruction Following with Alignment Self-Evaluation","date":"2024-09-02","arxiv_id":"2409.00935","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-judge-selective-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2409.00935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00935"}},"official":{"repos":["nusnlp/Self-J"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/does-alignment-tuning-really-break-llms","slug":"does-alignment-tuning-really-break-llms","title":"Does Alignment Tuning Really Break LLMs' Internal Confidence?","date":"2024-08-31","arxiv_id":"2409.00352","repositories_listed":1,"syntology":null},{"url":"/paper/scilitllm-how-to-adapt-llms-for-scientific","slug":"scilitllm-how-to-adapt-llms-for-scientific","title":"SciLitLLM: How to Adapt LLMs for Scientific Literature Understanding","date":"2024-08-28","arxiv_id":"2408.15545","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scilitllm-how-to-adapt-llms-for-scientific#ran","syntology_url":"https://syntology.ai/paper/2408.15545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15545"}},"official":{"repos":["dptech-corp/Uni-SMART"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/instruct-skillmix-a-powerful-pipeline-for-llm","slug":"instruct-skillmix-a-powerful-pipeline-for-llm","title":"Instruct-SkillMix: A Powerful Pipeline for LLM Instruction Tuning","date":"2024-08-27","arxiv_id":"2408.14774","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instruct-skillmix-a-powerful-pipeline-for-llm#ran","syntology_url":"https://syntology.ai/paper/2408.14774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.14774"}},"official":{"repos":["princeton-pli/Instruct-SkillMix"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/preference-guided-reflective-sampling-for","slug":"preference-guided-reflective-sampling-for","title":"Preference-Guided Reflective Sampling for Aligning Language Models","date":"2024-08-22","arxiv_id":"2408.12163","repositories_listed":1,"syntology":null},{"url":"/paper/ex3-automatic-novel-writing-by-extracting","slug":"ex3-automatic-novel-writing-by-extracting","title":"Ex3: Automatic Novel Writing by Extracting, Excelsior and Expanding","date":"2024-08-16","arxiv_id":"2408.08506","repositories_listed":1,"syntology":null},{"url":"/paper/llms-are-biased-towards-output-formats","slug":"llms-are-biased-towards-output-formats","title":"LLMs Are Biased Towards Output Formats! Systematically Evaluating and Mitigating Output Format Bias of LLMs","date":"2024-08-16","arxiv_id":"2408.08656","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llms-are-biased-towards-output-formats#ran","syntology_url":"https://syntology.ai/paper/2408.08656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08656"}},"official":{"repos":["dxlong2000/FormatEval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-and-modeling-correlations-in","slug":"bridging-and-modeling-correlations-in","title":"Bridging and Modeling Correlations in Pairwise Data for Direct Preference Optimization","date":"2024-08-14","arxiv_id":"2408.07471","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-and-modeling-correlations-in#ran","syntology_url":"https://syntology.ai/paper/2408.07471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07471"}},"official":{"repos":["YJiangcm/BMC"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ifship-a-large-vision-language-model-for","slug":"ifship-a-large-vision-language-model-for","title":"IFShip: Interpretable Fine-grained Ship Classification with Domain Knowledge-Enhanced Vision-Language Models","date":"2024-08-13","arxiv_id":"2408.06631","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-instruction-tuning-large","slug":"investigating-instruction-tuning-large","title":"Investigating Instruction Tuning Large Language Models on Graphs","date":"2024-08-10","arxiv_id":"2408.05457","repositories_listed":1,"syntology":null}],"record_sha256":"ad979243e3f303f02e46270519d3cb55150a806d53fd2e40d7b398df72b591a5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}