{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/language-modeling/papers/ran/5","list_of":"/task/language-modeling","task":"Language Modeling","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":5,"pages_in_order":19,"rows_per_page":100,"rows":[401,500],"of":1894,"counts":{"archive_papers_tagged":14182,"with_a_code_link":5620,"where_syntology_ran_a_sample":1894,"not_listed_spam_title":0,"listed":14182,"listed_where_code_ran":1894,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1580,"every_run_a_failure_of_syntologys_instrument":314,"listed_with_a_run_with_no_instrument_failure":1580,"listed_every_run_a_failure_of_syntologys_instrument":314,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/language-modeling/papers/ran/1","prev":"/task/language-modeling/papers/ran/4","next":"/task/language-modeling/papers/ran/6","papers":[{"url":"/paper/first-teach-a-reliable-large-language-model","slug":"first-teach-a-reliable-large-language-model","title":"FIRST: Teach A Reliable Large Language Model Through Efficient Trustworthy Distillation","date":"2024-08-22","arxiv_id":"2408.12168","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/first-teach-a-reliable-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2408.12168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12168"}},"official":{"repos":["shumkashun/first"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifashion-a-unified-vision-language-model","slug":"unifashion-a-unified-vision-language-model","title":"UniFashion: A Unified Vision-Language Model for Multimodal Fashion Retrieval and Generation","date":"2024-08-21","arxiv_id":"2408.11305","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":5,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unifashion-a-unified-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2408.11305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11305"}},"official":{"repos":["xiangyu-mm/unifashion"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/proteingpt-multimodal-llm-for-protein","slug":"proteingpt-multimodal-llm-for-protein","title":"ProteinGPT: Multimodal LLM for Protein Property Prediction and Structure Understanding","date":"2024-08-21","arxiv_id":"2408.11363","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/proteingpt-multimodal-llm-for-protein#ran","syntology_url":"https://syntology.ai/paper/2408.11363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11363"}},"official":{"repos":["proteingpt/proteingpt"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/focusllm-scaling-llm-s-context-by-parallel","slug":"focusllm-scaling-llm-s-context-by-parallel","title":"FocusLLM: Precise Understanding of Long Context by Dynamic Condensing","date":"2024-08-21","arxiv_id":"2408.11745","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/focusllm-scaling-llm-s-context-by-parallel#ran","syntology_url":"https://syntology.ai/paper/2408.11745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11745"}},"official":{"repos":["leezythu/focusllm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/approaching-deep-learning-through-the","slug":"approaching-deep-learning-through-the","title":"Approaching Deep Learning through the Spectral Dynamics of Weights","date":"2024-08-21","arxiv_id":"2408.11804","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/approaching-deep-learning-through-the#ran","syntology_url":"https://syntology.ai/paper/2408.11804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11804"}},"official":{"repos":["dyunis/spectral_dynamics"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/great-memory-shallow-reasoning-limits-of-k-nn","slug":"great-memory-shallow-reasoning-limits-of-k-nn","title":"Great Memory, Shallow Reasoning: Limits of $k$NN-LMs","date":"2024-08-21","arxiv_id":"2408.11815","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/great-memory-shallow-reasoning-limits-of-k-nn#ran","syntology_url":"https://syntology.ai/paper/2408.11815","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11815"}},"official":{"repos":["gsyfate/knnlm-limits"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/transfusion-predict-the-next-token-and","slug":"transfusion-predict-the-next-token-and","title":"Transfusion: Predict the Next Token and Diffuse Images with One Multi-Modal Model","date":"2024-08-20","arxiv_id":"2408.11039","repositories_listed":3,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":5,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 3 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/transfusion-predict-the-next-token-and#ran","syntology_url":"https://syntology.ai/paper/2408.11039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.11039"}},"official":null}},{"url":"/paper/blade-benchmarking-language-model-agents-for","slug":"blade-benchmarking-language-model-agents-for","title":"BLADE: Benchmarking Language Model Agents for Data-Driven Science","date":"2024-08-19","arxiv_id":"2408.09667","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/blade-benchmarking-language-model-agents-for#ran","syntology_url":"https://syntology.ai/paper/2408.09667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09667"}},"official":{"repos":["behavioral-data/blade"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/idea-enhancing-the-rule-learning-ability-of","slug":"idea-enhancing-the-rule-learning-ability-of","title":"IDEA: Enhancing the Rule Learning Ability of Large Language Model Agent through Induction, Deduction, and Abduction","date":"2024-08-19","arxiv_id":"2408.10455","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/idea-enhancing-the-rule-learning-ability-of#ran","syntology_url":"https://syntology.ai/paper/2408.10455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10455"}},"official":{"repos":["kaiyuhe998/rulearn_idea"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hiagent-hierarchical-working-memory","slug":"hiagent-hierarchical-working-memory","title":"HiAgent: Hierarchical Working Memory Management for Solving Long-Horizon Agent Tasks with Large Language Model","date":"2024-08-18","arxiv_id":"2408.09559","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hiagent-hierarchical-working-memory#ran","syntology_url":"https://syntology.ai/paper/2408.09559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09559"}},"official":{"repos":["hiagent2024/hiagent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ecg-chat-a-large-ecg-language-model-for","slug":"ecg-chat-a-large-ecg-language-model-for","title":"ECG-Chat: A Large ECG-Language Model for Cardiac Disease Diagnosis","date":"2024-08-16","arxiv_id":"2408.08849","repositories_listed":1,"syntology":{"n":14,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":14,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/ecg-chat-a-large-ecg-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2408.08849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08849"}},"official":{"repos":["YubaoZhao/ECG-Chat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/text2bim-generating-building-models-using-a","slug":"text2bim-generating-building-models-using-a","title":"Text2BIM: Generating Building Models Using a Large Language Model-based Multi-Agent Framework","date":"2024-08-15","arxiv_id":"2408.08054","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/text2bim-generating-building-models-using-a#ran","syntology_url":"https://syntology.ai/paper/2408.08054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08054"}},"official":{"repos":["dcy0577/Text2BIM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-prover-v1-5-harnessing-proof","slug":"deepseek-prover-v1-5-harnessing-proof","title":"DeepSeek-Prover-V1.5: Harnessing Proof Assistant Feedback for Reinforcement Learning and Monte-Carlo Tree Search","date":"2024-08-15","arxiv_id":"2408.08152","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepseek-prover-v1-5-harnessing-proof#ran","syntology_url":"https://syntology.ai/paper/2408.08152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08152"}},"official":{"repos":["deepseek-ai/deepseek-prover-v1.5"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/seeing-and-understanding-bridging-vision-with","slug":"seeing-and-understanding-bridging-vision-with","title":"ChemVLM: Exploring the Power of Multimodal Large Language Models in Chemistry Area","date":"2024-08-14","arxiv_id":"2408.07246","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/seeing-and-understanding-bridging-vision-with#ran","syntology_url":"https://syntology.ai/paper/2408.07246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07246"}},"official":{"repos":["AI4Chem/ChemVlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-agent-based-on-large-language-model","slug":"causal-agent-based-on-large-language-model","title":"Causal Agent based on Large Language Model","date":"2024-08-13","arxiv_id":"2408.06849","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/causal-agent-based-on-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2408.06849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06849"}},"official":{"repos":["kairong-han/causal_agent"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dyg-mamba-continuous-state-space-modeling-on","slug":"dyg-mamba-continuous-state-space-modeling-on","title":"DyG-Mamba: Continuous State Space Modeling on Dynamic Graphs","date":"2024-08-13","arxiv_id":"2408.06966","repositories_listed":0,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dyg-mamba-continuous-state-space-modeling-on#ran","syntology_url":"https://syntology.ai/paper/2408.06966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06966"}},"official":null}},{"url":"/paper/on-effects-of-steering-latent-representation","slug":"on-effects-of-steering-latent-representation","title":"On Effects of Steering Latent Representation for Large Language Model Unlearning","date":"2024-08-12","arxiv_id":"2408.06223","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-effects-of-steering-latent-representation#ran","syntology_url":"https://syntology.ai/paper/2408.06223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06223"}},"official":{"repos":["RebelsNLU-jaist/llm-unlearning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-ai-scientist-towards-fully-automated-open","slug":"the-ai-scientist-towards-fully-automated-open","title":"The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery","date":"2024-08-12","arxiv_id":"2408.06292","repositories_listed":2,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-ai-scientist-towards-fully-automated-open#ran","syntology_url":"https://syntology.ai/paper/2408.06292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.06292"}},"official":{"repos":["sakanaai/ai-scientist"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/unibench-visual-reasoning-requires-rethinking","slug":"unibench-visual-reasoning-requires-rethinking","title":"UniBench: Visual Reasoning Requires Rethinking Vision-Language Beyond Scaling","date":"2024-08-09","arxiv_id":"2408.04810","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/unibench-visual-reasoning-requires-rethinking#ran","syntology_url":"https://syntology.ai/paper/2408.04810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04810"}},"official":{"repos":["facebookresearch/unibench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/medical-graph-rag-towards-safe-medical-large","slug":"medical-graph-rag-towards-safe-medical-large","title":"Medical Graph RAG: Towards Safe Medical Large Language Model via Graph Retrieval-Augmented Generation","date":"2024-08-08","arxiv_id":"2408.04187","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/medical-graph-rag-towards-safe-medical-large#ran","syntology_url":"https://syntology.ai/paper/2408.04187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04187"}},"official":{"repos":["medicinetoken/medical-graph-rag"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-guided-language-modeling","slug":"diffusion-guided-language-modeling","title":"Diffusion Guided Language Modeling","date":"2024-08-08","arxiv_id":"2408.04220","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":11,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 2 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diffusion-guided-language-modeling#ran","syntology_url":"https://syntology.ai/paper/2408.04220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04220"}},"official":{"repos":["justinlovelace/diffusion-guided-lm"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/1-5-pints-technical-report-pretraining-in","slug":"1-5-pints-technical-report-pretraining-in","title":"1.5-Pints Technical Report: Pretraining in Days, Not Months -- Your Language Model Thrives on Quality Data","date":"2024-08-07","arxiv_id":"2408.03506","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/1-5-pints-technical-report-pretraining-in#ran","syntology_url":"https://syntology.ai/paper/2408.03506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03506"}},"official":{"repos":["Pints-AI/1.5-Pints"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-child-directed-speech-effective-training","slug":"is-child-directed-speech-effective-training","title":"Is Child-Directed Speech Effective Training Data for Language Models?","date":"2024-08-07","arxiv_id":"2408.03617","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/is-child-directed-speech-effective-training#ran","syntology_url":"https://syntology.ai/paper/2408.03617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03617"}},"official":{"repos":["styfeng/tinydialogues"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ullme-a-unified-framework-for-large-language","slug":"ullme-a-unified-framework-for-large-language","title":"ULLME: A Unified Framework for Large Language Model Embeddings with Generation-Augmented Learning","date":"2024-08-06","arxiv_id":"2408.03402","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ullme-a-unified-framework-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2408.03402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.03402"}},"official":{"repos":["nlp-uoregon/ullme"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citekit-a-modular-toolkit-for-large-language","slug":"citekit-a-modular-toolkit-for-large-language","title":"Citekit: A Modular Toolkit for Large Language Model Citation Generation","date":"2024-08-06","arxiv_id":"2408.04662","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/citekit-a-modular-toolkit-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2408.04662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04662"}},"official":{"repos":["sjj1017/citekit"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-stability-a-detailed-analysis-with-some","slug":"llm-stability-a-detailed-analysis-with-some","title":"Non-Determinism of \"Deterministic\" LLM Settings","date":"2024-08-06","arxiv_id":"2408.04667","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-stability-a-detailed-analysis-with-some#ran","syntology_url":"https://syntology.ai/paper/2408.04667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04667"}},"official":{"repos":["breckbaldwin/llm-stability","Comcast/llm-stability"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-02544","slug":"2408-02544","title":"Caution for the Environment: Multimodal Agents are Susceptible to Environmental Distractions","date":"2024-08-05","arxiv_id":"2408.02544","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2408-02544#ran","syntology_url":"https://syntology.ai/paper/2408.02544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.02544"}},"official":{"repos":["xbmxb/EnvDistraction"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00764","slug":"2408-00764","title":"AgentGen: Enhancing Planning Abilities for Large Language Model based Agent via Environment and Task Generation","date":"2024-08-01","arxiv_id":"2408.00764","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/2408-00764#ran","syntology_url":"https://syntology.ai/paper/2408.00764","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00764"}},"official":{"repos":["lazychih114/AgentGen-Reproduction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/2407-21757","slug":"2407-21757","title":"Learning Video Context as Interleaved Multimodal Sequences","date":"2024-07-31","arxiv_id":"2407.21757","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2407-21757#ran","syntology_url":"https://syntology.ai/paper/2407.21757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21757"}},"official":{"repos":["showlab/movieseq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-llama-3-herd-of-models","slug":"the-llama-3-herd-of-models","title":"The Llama 3 Herd of Models","date":"2024-07-31","arxiv_id":"2407.21783","repositories_listed":5,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-llama-3-herd-of-models#ran","syntology_url":"https://syntology.ai/paper/2407.21783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21783"}},"official":null}},{"url":"/paper/2408-00113","slug":"2408-00113","title":"Measuring Progress in Dictionary Learning for Language Model Interpretability with Board Game Models","date":"2024-07-31","arxiv_id":"2408.00113","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-00113#ran","syntology_url":"https://syntology.ai/paper/2408.00113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.00113"}},"official":{"repos":["adamkarvonen/SAE_BoardGameEval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimus-0-3-using-large-language-models-to","slug":"optimus-0-3-using-large-language-models-to","title":"OptiMUS-0.3: Using Large Language Models to Model and Solve Optimization Problems at Scale","date":"2024-07-29","arxiv_id":"2407.19633","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/optimus-0-3-using-large-language-models-to#ran","syntology_url":"https://syntology.ai/paper/2407.19633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19633"}},"official":null}},{"url":"/paper/demystifying-verbatim-memorization-in-large","slug":"demystifying-verbatim-memorization-in-large","title":"Demystifying Verbatim Memorization in Large Language Models","date":"2024-07-25","arxiv_id":"2407.17817","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/demystifying-verbatim-memorization-in-large#ran","syntology_url":"https://syntology.ai/paper/2407.17817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17817"}},"official":{"repos":["explanare/verbatim-memorization"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-scaling-trends-in-llm-robustness","slug":"exploring-scaling-trends-in-llm-robustness","title":"Scaling Trends in Language Model Robustness","date":"2024-07-25","arxiv_id":"2407.18213","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-scaling-trends-in-llm-robustness#ran","syntology_url":"https://syntology.ai/paper/2407.18213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18213"}},"official":{"repos":["AlignmentResearch/scaling-llm-robustness-paper"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/train-attention-meta-learning-where-to-focus","slug":"train-attention-meta-learning-where-to-focus","title":"Train-Attention: Meta-Learning Where to Focus in Continual Knowledge Learning","date":"2024-07-24","arxiv_id":"2407.16920","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/train-attention-meta-learning-where-to-focus#ran","syntology_url":"https://syntology.ai/paper/2407.16920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16920"}},"official":{"repos":["ybseo-academy/TAALM"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/inf-llava-dual-perspective-perception-for","slug":"inf-llava-dual-perspective-perception-for","title":"INF-LLaVA: Dual-perspective Perception for High-Resolution Multimodal Large Language Model","date":"2024-07-23","arxiv_id":"2407.16198","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inf-llava-dual-perspective-perception-for#ran","syntology_url":"https://syntology.ai/paper/2407.16198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16198"}},"official":{"repos":["weihuanglin/inf-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/prealign-boosting-cross-lingual-transfer-by","slug":"prealign-boosting-cross-lingual-transfer-by","title":"PreAlign: Boosting Cross-Lingual Transfer by Early Establishment of Multilingual Alignment","date":"2024-07-23","arxiv_id":"2407.16222","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/prealign-boosting-cross-lingual-transfer-by#ran","syntology_url":"https://syntology.ai/paper/2407.16222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16222"}},"official":{"repos":["saltychtao/prealign"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/amongagents-evaluating-large-language-models","slug":"amongagents-evaluating-large-language-models","title":"AMONGAGENTS: Evaluating Large Language Models in the Interactive Text-Based Social Deduction Game","date":"2024-07-23","arxiv_id":"2407.16521","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/amongagents-evaluating-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.16521","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16521"}},"official":{"repos":["cyzus/among-agents"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tlcr-token-level-continuous-reward-for-fine","slug":"tlcr-token-level-continuous-reward-for-fine","title":"TLCR: Token-Level Continuous Reward for Fine-grained Reinforcement Learning from Human Feedback","date":"2024-07-23","arxiv_id":"2407.16574","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tlcr-token-level-continuous-reward-for-fine#ran","syntology_url":"https://syntology.ai/paper/2407.16574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16574"}},"official":{"repos":["esyoon7/rlhf-tlcr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lawma-the-power-of-specialization-for-legal","slug":"lawma-the-power-of-specialization-for-legal","title":"Lawma: The Power of Specialization for Legal Tasks","date":"2024-07-23","arxiv_id":"2407.16615","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lawma-the-power-of-specialization-for-legal#ran","syntology_url":"https://syntology.ai/paper/2407.16615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16615"}},"official":null}},{"url":"/paper/adaclip-adapting-clip-with-hybrid-learnable","slug":"adaclip-adapting-clip-with-hybrid-learnable","title":"AdaCLIP: Adapting CLIP with Hybrid Learnable Prompts for Zero-Shot Anomaly Detection","date":"2024-07-22","arxiv_id":"2407.15795","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":10,"n_ran_checked":16,"n_instrument":4,"n_unverified":8,"n_honours":0,"n_violates":1,"n_no_contract":15,"n_pointer_only":4,"phrase":"20 ran (of which 10 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 1 violated, 15 with no contract checked; 4 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/adaclip-adapting-clip-with-hybrid-learnable#ran","syntology_url":"https://syntology.ai/paper/2407.15795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15795"}},"official":{"repos":["caoyunkang/adaclip"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":10,"n_ran_no_instrument_failure":16,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/dmel-speech-tokenization-made-simple","slug":"dmel-speech-tokenization-made-simple","title":"dMel: Speech Tokenization made Simple","date":"2024-07-22","arxiv_id":"2407.15835","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dmel-speech-tokenization-made-simple#ran","syntology_url":"https://syntology.ai/paper/2407.15835","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15835"}},"official":{"repos":["apple/dmel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slowfast-llava-a-strong-training-free","slug":"slowfast-llava-a-strong-training-free","title":"SlowFast-LLaVA: A Strong Training-Free Baseline for Video Large Language Models","date":"2024-07-22","arxiv_id":"2407.15841","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/slowfast-llava-a-strong-training-free#ran","syntology_url":"https://syntology.ai/paper/2407.15841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15841"}},"official":{"repos":["apple/ml-slowfast-llava"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-for-verilog-generation","slug":"large-language-model-for-verilog-generation","title":"Large Language Model for Verilog Generation with Code-Structure-Guided Reinforcement Learning","date":"2024-07-21","arxiv_id":"2407.18271","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-model-for-verilog-generation#ran","syntology_url":"https://syntology.ai/paper/2407.18271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18271"}},"official":{"repos":["CatIIIIIIII/veriseek"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rag-qa-arena-evaluating-domain-robustness-for","slug":"rag-qa-arena-evaluating-domain-robustness-for","title":"RAG-QA Arena: Evaluating Domain Robustness for Long-form Retrieval Augmented Question Answering","date":"2024-07-19","arxiv_id":"2407.13998","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rag-qa-arena-evaluating-domain-robustness-for#ran","syntology_url":"https://syntology.ai/paper/2407.13998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13998"}},"official":{"repos":["awslabs/rag-qa-arena"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longhorn-state-space-models-are-amortized","slug":"longhorn-state-space-models-are-amortized","title":"Longhorn: State Space Models are Amortized Online Learners","date":"2024-07-19","arxiv_id":"2407.14207","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longhorn-state-space-models-are-amortized#ran","syntology_url":"https://syntology.ai/paper/2407.14207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14207"}},"official":{"repos":["Cranial-XIX/longhorn"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/t2v-compbench-a-comprehensive-benchmark-for","slug":"t2v-compbench-a-comprehensive-benchmark-for","title":"T2V-CompBench: A Comprehensive Benchmark for Compositional Text-to-video Generation","date":"2024-07-19","arxiv_id":"2407.14505","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/t2v-compbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2407.14505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14505"}},"official":{"repos":["KaiyueSun98/T2V-CompBench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-the-generalization-and-reliability","slug":"analyzing-the-generalization-and-reliability","title":"Analyzing the Generalization and Reliability of Steering Vectors","date":"2024-07-17","arxiv_id":"2407.12404","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/analyzing-the-generalization-and-reliability#ran","syntology_url":"https://syntology.ai/paper/2407.12404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12404"}},"official":{"repos":["dtch1997/steering-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/patch-level-training-for-large-language","slug":"patch-level-training-for-large-language","title":"Beyond Next Token Prediction: Patch-Level Training for Large Language Models","date":"2024-07-17","arxiv_id":"2407.12665","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/patch-level-training-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2407.12665","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12665"}},"official":{"repos":["shaochenze/patchtrain"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lami-detr-open-vocabulary-detection-with","slug":"lami-detr-open-vocabulary-detection-with","title":"LaMI-DETR: Open-Vocabulary Detection with Language Model Instruction","date":"2024-07-16","arxiv_id":"2407.11335","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lami-detr-open-vocabulary-detection-with#ran","syntology_url":"https://syntology.ai/paper/2407.11335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11335"}},"official":{"repos":["eternaldolphin/lami-detr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/think-on-graph-2-0-deep-and-interpretable","slug":"think-on-graph-2-0-deep-and-interpretable","title":"Think-on-Graph 2.0: Deep and Faithful Large Language Model Reasoning with Knowledge-guided Retrieval Augmented Generation","date":"2024-07-15","arxiv_id":"2407.10805","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 1 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/think-on-graph-2-0-deep-and-interpretable#ran","syntology_url":"https://syntology.ai/paper/2407.10805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10805"}},"official":{"repos":["idea-finai/tog-2"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-actionable-framework-for-assessing-bias","slug":"an-actionable-framework-for-assessing-bias","title":"An Actionable Framework for Assessing Bias and Fairness in Large Language Model Use Cases","date":"2024-07-15","arxiv_id":"2407.10853","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-actionable-framework-for-assessing-bias#ran","syntology_url":"https://syntology.ai/paper/2407.10853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10853"}},"official":{"repos":["cvs-health/langfair"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/leanquant-accurate-large-language-model","slug":"leanquant-accurate-large-language-model","title":"LeanQuant: Accurate Large Language Model Quantization with Loss-Error-Aware Grid","date":"2024-07-14","arxiv_id":"2407.10032","repositories_listed":0,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/leanquant-accurate-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2407.10032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10032"}},"official":null}},{"url":"/paper/practical-unlearning-for-large-language","slug":"practical-unlearning-for-large-language","title":"On Large Language Model Continual Unlearning","date":"2024-07-14","arxiv_id":"2407.10223","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/practical-unlearning-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2407.10223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10223"}},"official":{"repos":["gcyzsl/o3-llm-unlearning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/benchmarking-language-model-creativity-a-case","slug":"benchmarking-language-model-creativity-a-case","title":"Benchmarking Language Model Creativity: A Case Study on Code Generation","date":"2024-07-12","arxiv_id":"2407.09007","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-language-model-creativity-a-case#ran","syntology_url":"https://syntology.ai/paper/2407.09007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09007"}},"official":{"repos":["JHU-CLSP/NeoCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-diffusion-behaviors-with-q-functions","slug":"aligning-diffusion-behaviors-with-q-functions","title":"Aligning Diffusion Behaviors with Q-functions for Efficient Continuous Control","date":"2024-07-12","arxiv_id":"2407.09024","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":12,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/aligning-diffusion-behaviors-with-q-functions#ran","syntology_url":"https://syntology.ai/paper/2407.09024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09024"}},"official":{"repos":["thu-ml/efficient-diffusion-alignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gofa-a-generative-one-for-all-model-for-joint","slug":"gofa-a-generative-one-for-all-model-for-joint","title":"GOFA: A Generative One-For-All Model for Joint Graph Language Modeling","date":"2024-07-12","arxiv_id":"2407.09709","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gofa-a-generative-one-for-all-model-for-joint#ran","syntology_url":"https://syntology.ai/paper/2407.09709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09709"}},"official":{"repos":["jiaruifeng/gofa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/explore-the-potential-of-clip-for-training","slug":"explore-the-potential-of-clip-for-training","title":"Explore the Potential of CLIP for Training-Free Open Vocabulary Semantic Segmentation","date":"2024-07-11","arxiv_id":"2407.08268","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/explore-the-potential-of-clip-for-training#ran","syntology_url":"https://syntology.ai/paper/2407.08268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08268"}},"official":{"repos":["leaves162/cliptrase"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/seed-story-multimodal-long-story-generation","slug":"seed-story-multimodal-long-story-generation","title":"SEED-Story: Multimodal Long Story Generation with Large Language Model","date":"2024-07-11","arxiv_id":"2407.08683","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":8,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":17,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/seed-story-multimodal-long-story-generation#ran","syntology_url":"https://syntology.ai/paper/2407.08683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08683"}},"official":{"repos":["tencentarc/seed-story"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ida-vlm-towards-movie-understanding-via-id","slug":"ida-vlm-towards-movie-understanding-via-id","title":"IDA-VLM: Towards Movie Understanding via ID-Aware Large Vision-Language Model","date":"2024-07-10","arxiv_id":"2407.07577","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ida-vlm-towards-movie-understanding-via-id#ran","syntology_url":"https://syntology.ai/paper/2407.07577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07577"}},"official":{"repos":["jiyt17/ida-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/paligemma-a-versatile-3b-vlm-for-transfer","slug":"paligemma-a-versatile-3b-vlm-for-transfer","title":"PaliGemma: A versatile 3B VLM for transfer","date":"2024-07-10","arxiv_id":"2407.07726","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/paligemma-a-versatile-3b-vlm-for-transfer#ran","syntology_url":"https://syntology.ai/paper/2407.07726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07726"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cola-conditional-dropout-and-language-driven","slug":"cola-conditional-dropout-and-language-driven","title":"CoLA: Conditional Dropout and Language-driven Robust Dual-modal Salient Object Detection","date":"2024-07-09","arxiv_id":"2407.06780","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cola-conditional-dropout-and-language-driven#ran","syntology_url":"https://syntology.ai/paper/2407.06780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06780"}},"official":{"repos":["ssecv/CoLA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-self-instruct-synthetic-abstract","slug":"multimodal-self-instruct-synthetic-abstract","title":"Multimodal Self-Instruct: Synthetic Abstract Image and Visual Reasoning Instruction Using Language Model","date":"2024-07-09","arxiv_id":"2407.07053","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-self-instruct-synthetic-abstract#ran","syntology_url":"https://syntology.ai/paper/2407.07053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07053"}},"official":{"repos":["zwq2018/multi-modal-self-instruct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/copybench-measuring-literal-and-non-literal","slug":"copybench-measuring-literal-and-non-literal","title":"CopyBench: Measuring Literal and Non-Literal Reproduction of Copyright-Protected Text in Language Model Generation","date":"2024-07-09","arxiv_id":"2407.07087","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/copybench-measuring-literal-and-non-literal#ran","syntology_url":"https://syntology.ai/paper/2407.07087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07087"}},"official":{"repos":["chentong0/copy-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fbi-llm-scaling-up-fully-binarized-llms-from","slug":"fbi-llm-scaling-up-fully-binarized-llms-from","title":"FBI-LLM: Scaling Up Fully Binarized LLMs from Scratch via Autoregressive Distillation","date":"2024-07-09","arxiv_id":"2407.07093","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/fbi-llm-scaling-up-fully-binarized-llms-from#ran","syntology_url":"https://syntology.ai/paper/2407.07093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07093"}},"official":{"repos":["liqunma/fbi-llm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-retrieval-based-language-models-with","slug":"scaling-retrieval-based-language-models-with","title":"Scaling Retrieval-Based Language Models with a Trillion-Token Datastore","date":"2024-07-09","arxiv_id":"2407.12854","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scaling-retrieval-based-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2407.12854","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12854"}},"official":{"repos":["rulinshao/retrieval-scaling"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-speeding-up-language-model-evaluation","slug":"on-speeding-up-language-model-evaluation","title":"On Speeding Up Language Model Evaluation","date":"2024-07-08","arxiv_id":"2407.06172","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-speeding-up-language-model-evaluation#ran","syntology_url":"https://syntology.ai/paper/2407.06172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06172"}},"official":null}},{"url":"/paper/debunc-mitigating-hallucinations-in-large","slug":"debunc-mitigating-hallucinations-in-large","title":"DebUnc: Improving Large Language Model Agent Communication With Uncertainty Metrics","date":"2024-07-08","arxiv_id":"2407.06426","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/debunc-mitigating-hallucinations-in-large#ran","syntology_url":"https://syntology.ai/paper/2407.06426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06426"}},"official":{"repos":["lukeyoffe/debunc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-single-transformer-for-scalable-vision","slug":"a-single-transformer-for-scalable-vision","title":"SOLO: A Single Transformer for Scalable Vision-Language Modeling","date":"2024-07-08","arxiv_id":"2407.06438","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-single-transformer-for-scalable-vision#ran","syntology_url":"https://syntology.ai/paper/2407.06438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06438"}},"official":{"repos":["yangyi-chen/solo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-encode-collaborative-signals","slug":"language-models-encode-collaborative-signals","title":"Language Representations Can be What Recommenders Need: Findings and Potentials","date":"2024-07-07","arxiv_id":"2407.05441","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-encode-collaborative-signals#ran","syntology_url":"https://syntology.ai/paper/2407.05441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05441"}},"official":{"repos":["lehengthu/alpharec"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-hallucination-detection-through","slug":"enhancing-hallucination-detection-through","title":"Enhancing Hallucination Detection through Perturbation-Based Synthetic Data Generation in System Responses","date":"2024-07-07","arxiv_id":"2407.05474","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-hallucination-detection-through#ran","syntology_url":"https://syntology.ai/paper/2407.05474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05474"}},"official":{"repos":["asappresearch/halugen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/just-read-twice-closing-the-recall-gap-for","slug":"just-read-twice-closing-the-recall-gap-for","title":"Just read twice: closing the recall gap for recurrent language models","date":"2024-07-07","arxiv_id":"2407.05483","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/just-read-twice-closing-the-recall-gap-for#ran","syntology_url":"https://syntology.ai/paper/2407.05483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05483"}},"official":{"repos":["HazyResearch/prefix-linear-attention"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-perplexity-multi-dimensional-safety","slug":"beyond-perplexity-multi-dimensional-safety","title":"Beyond Perplexity: Multi-dimensional Safety Evaluation of LLM Compression","date":"2024-07-06","arxiv_id":"2407.04965","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":16,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/beyond-perplexity-multi-dimensional-safety#ran","syntology_url":"https://syntology.ai/paper/2407.04965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04965"}},"official":{"repos":["zhichaoxu-shufe/beyond-perplexity-compression-safety-eval"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/shine-saliency-aware-hierarchical-negative","slug":"shine-saliency-aware-hierarchical-negative","title":"SHINE: Saliency-aware HIerarchical NEgative Ranking for Compositional Temporal Grounding","date":"2024-07-06","arxiv_id":"2407.05118","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/shine-saliency-aware-hierarchical-negative#ran","syntology_url":"https://syntology.ai/paper/2407.05118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05118"}},"official":{"repos":["zxccade/shine"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-good-medical-coders","slug":"large-language-models-are-good-medical-coders","title":"Large language models are good medical coders, if provided with tools","date":"2024-07-06","arxiv_id":"2407.12849","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-are-good-medical-coders#ran","syntology_url":"https://syntology.ai/paper/2407.12849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12849"}},"official":{"repos":["ainativehealth/goodmedicalcoder"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crafting-large-language-models-for-enhanced","slug":"crafting-large-language-models-for-enhanced","title":"Crafting Large Language Models for Enhanced Interpretability","date":"2024-07-05","arxiv_id":"2407.04307","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/crafting-large-language-models-for-enhanced#ran","syntology_url":"https://syntology.ai/paper/2407.04307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04307"}},"official":null}},{"url":"/paper/minigpt-med-large-language-model-as-a-general","slug":"minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","arxiv_id":"2407.04106","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-med-large-language-model-as-a-general#ran","syntology_url":"https://syntology.ai/paper/2407.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04106"}},"official":{"repos":["vision-cair/minigpt-med"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mixture-of-a-million-experts","slug":"mixture-of-a-million-experts","title":"Mixture of A Million Experts","date":"2024-07-04","arxiv_id":"2407.04153","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mixture-of-a-million-experts#ran","syntology_url":"https://syntology.ai/paper/2407.04153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04153"}},"official":null}},{"url":"/paper/on-the-client-preference-of-llm-fine-tuning","slug":"on-the-client-preference-of-llm-fine-tuning","title":"Towards Federated RLHF with Aggregated Client Preference for LLMs","date":"2024-07-03","arxiv_id":"2407.03038","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-client-preference-of-llm-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2407.03038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03038"}},"official":null}},{"url":"/paper/internlm-xcomposer-2-5-a-versatile-large","slug":"internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","arxiv_id":"2407.03320","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internlm-xcomposer-2-5-a-versatile-large#ran","syntology_url":"https://syntology.ai/paper/2407.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03320"}},"official":{"repos":["internlm/internlm-xcomposer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/helpful-assistant-or-fruitful-facilitator","slug":"helpful-assistant-or-fruitful-facilitator","title":"Helpful assistant or fruitful facilitator? Investigating how personas affect language model behavior","date":"2024-07-02","arxiv_id":"2407.02099","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/helpful-assistant-or-fruitful-facilitator#ran","syntology_url":"https://syntology.ai/paper/2407.02099","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02099"}},"official":{"repos":["peluz/persona-behavior"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multilingual-trolley-problems-for-language","slug":"multilingual-trolley-problems-for-language","title":"Language Model Alignment in Multilingual Trolley Problems","date":"2024-07-02","arxiv_id":"2407.02273","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multilingual-trolley-problems-for-language#ran","syntology_url":"https://syntology.ai/paper/2407.02273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02273"}},"official":{"repos":["causalNLP/moralmachine","causalnlp/multitp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-involuntary-truth","slug":"large-language-models-are-involuntary-truth","title":"Large Language Models Are Involuntary Truth-Tellers: Exploiting Fallacy Failure for Jailbreak Attacks","date":"2024-07-01","arxiv_id":"2407.00869","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/large-language-models-are-involuntary-truth#ran","syntology_url":"https://syntology.ai/paper/2407.00869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00869"}},"official":{"repos":["Yue-LLM-Pit/FFA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ibsen-director-actor-agent-collaboration-for","slug":"ibsen-director-actor-agent-collaboration-for","title":"IBSEN: Director-Actor Agent Collaboration for Controllable and Interactive Drama Script Generation","date":"2024-07-01","arxiv_id":"2407.01093","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ibsen-director-actor-agent-collaboration-for#ran","syntology_url":"https://syntology.ai/paper/2407.01093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01093"}},"official":{"repos":["OpenDFM/ibsen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tree-search-for-language-model-agents","slug":"tree-search-for-language-model-agents","title":"Tree Search for Language Model Agents","date":"2024-07-01","arxiv_id":"2407.01476","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tree-search-for-language-model-agents#ran","syntology_url":"https://syntology.ai/paper/2407.01476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01476"}},"official":null}},{"url":"/paper/regmix-data-mixture-as-regression-for","slug":"regmix-data-mixture-as-regression-for","title":"RegMix: Data Mixture as Regression for Language Model Pre-training","date":"2024-07-01","arxiv_id":"2407.01492","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regmix-data-mixture-as-regression-for#ran","syntology_url":"https://syntology.ai/paper/2407.01492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01492"}},"official":{"repos":["sail-sg/regmix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crab-cross-environment-agent-benchmark-for","slug":"crab-cross-environment-agent-benchmark-for","title":"CRAB: Cross-environment Agent Benchmark for Multimodal Language Model Agents","date":"2024-07-01","arxiv_id":"2407.01511","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crab-cross-environment-agent-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2407.01511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01511"}},"official":{"repos":["camel-ai/crab"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meerkat-audio-visual-large-language-model-for","slug":"meerkat-audio-visual-large-language-model-for","title":"Meerkat: Audio-Visual Large Language Model for Grounding in Space and Time","date":"2024-07-01","arxiv_id":"2407.01851","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/meerkat-audio-visual-large-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2407.01851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01851"}},"official":{"repos":["schowdhury671/meerkat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/autoflow-automated-workflow-generation-for","slug":"autoflow-automated-workflow-generation-for","title":"AutoFlow: Automated Workflow Generation for Large Language Model Agents","date":"2024-07-01","arxiv_id":"2407.12821","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autoflow-automated-workflow-generation-for#ran","syntology_url":"https://syntology.ai/paper/2407.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12821"}},"official":{"repos":["agiresearch/autoflow"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-formal-mathematics-from-intrinsic","slug":"learning-formal-mathematics-from-intrinsic","title":"Learning Formal Mathematics From Intrinsic Motivation","date":"2024-06-30","arxiv_id":"2407.00695","repositories_listed":2,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/learning-formal-mathematics-from-intrinsic#ran","syntology_url":"https://syntology.ai/paper/2407.00695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00695"}},"official":{"repos":["gpoesia/minimo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/yulan-an-open-source-large-language-model","slug":"yulan-an-open-source-large-language-model","title":"YuLan: An Open-source Large Language Model","date":"2024-06-28","arxiv_id":"2406.19853","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/yulan-an-open-source-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2406.19853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19853"}},"official":{"repos":["ruc-gsai/yulan-chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evf-sam-early-vision-language-fusion-for-text","slug":"evf-sam-early-vision-language-fusion-for-text","title":"EVF-SAM: Early Vision-Language Fusion for Text-Prompted Segment Anything Model","date":"2024-06-28","arxiv_id":"2406.20076","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evf-sam-early-vision-language-fusion-for-text#ran","syntology_url":"https://syntology.ai/paper/2406.20076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20076"}},"official":{"repos":["hustvl/evf-sam"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/scaling-synthetic-data-creation-with","slug":"scaling-synthetic-data-creation-with","title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","date":"2024-06-28","arxiv_id":"2406.20094","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-synthetic-data-creation-with#ran","syntology_url":"https://syntology.ai/paper/2406.20094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20094"}},"official":{"repos":["tencent-ailab/persona-hub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/decoding-time-language-model-alignment-with","slug":"decoding-time-language-model-alignment-with","title":"Decoding-Time Language Model Alignment with Multiple Objectives","date":"2024-06-27","arxiv_id":"2406.18853","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/decoding-time-language-model-alignment-with#ran","syntology_url":"https://syntology.ai/paper/2406.18853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18853"}},"official":{"repos":["srzer/mod"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/efficacy-of-language-model-self-play-in-non","slug":"efficacy-of-language-model-self-play-in-non","title":"Efficacy of Language Model Self-Play in Non-Zero-Sum Games","date":"2024-06-27","arxiv_id":"2406.18872","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficacy-of-language-model-self-play-in-non#ran","syntology_url":"https://syntology.ai/paper/2406.18872","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18872"}},"official":{"repos":["nickatomlin/lm-selfplay"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robouniview-visual-language-model-with","slug":"robouniview-visual-language-model-with","title":"RoboUniView: Visual-Language Model with Unified View Representation for Robotic Manipulation","date":"2024-06-27","arxiv_id":"2406.18977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robouniview-visual-language-model-with#ran","syntology_url":"https://syntology.ai/paper/2406.18977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18977"}},"official":{"repos":["liufanfanlff/robouniview"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-refer-and-ground-multimodal-large-language","slug":"a-refer-and-ground-multimodal-large-language","title":"A Refer-and-Ground Multimodal Large Language Model for Biomedicine","date":"2024-06-26","arxiv_id":"2406.18146","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-refer-and-ground-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.18146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18146"}},"official":{"repos":["shawnhuang497/bird"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/themis-towards-flexible-and-interpretable-nlg","slug":"themis-towards-flexible-and-interpretable-nlg","title":"Themis: A Reference-free NLG Evaluation Language Model with Flexibility and Interpretability","date":"2024-06-26","arxiv_id":"2406.18365","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/themis-towards-flexible-and-interpretable-nlg#ran","syntology_url":"https://syntology.ai/paper/2406.18365","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18365"}},"official":{"repos":["PKU-ONELab/Themis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-we-trust-the-performance-evaluation-of","slug":"can-we-trust-the-performance-evaluation-of","title":"Can We Trust the Performance Evaluation of Uncertainty Estimation Methods in Text Summarization?","date":"2024-06-25","arxiv_id":"2406.17274","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-we-trust-the-performance-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2406.17274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17274"}},"official":{"repos":["he159ok/benchmark-of-uncertainty-estimation-methods-in-text-summarization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"9efcfc61c7d2387e97ba5999a7a4f5970faad4b4a29a6b44d296150eb44b4603","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}