{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/large-language-model/papers/ran/5","list_of":"/task/large-language-model","task":"Large Language Model","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":5,"pages_in_order":9,"rows_per_page":100,"rows":[401,500],"of":801,"counts":{"archive_papers_tagged":6097,"with_a_code_link":2250,"where_syntology_ran_a_sample":801,"not_listed_spam_title":0,"listed":6097,"listed_where_code_ran":801,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":683,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":683,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/large-language-model/papers/ran/1","prev":"/task/large-language-model/papers/ran/4","next":"/task/large-language-model/papers/ran/6","papers":[{"url":"/paper/openbias-open-set-bias-detection-in-text-to","slug":"openbias-open-set-bias-detection-in-text-to","title":"OpenBias: Open-set Bias Detection in Text-to-Image Generative Models","date":"2024-04-11","arxiv_id":"2404.07990","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openbias-open-set-bias-detection-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2404.07990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07990"}},"official":{"repos":["picsart-ai-research/openbias"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/freeeval-a-modular-framework-for-trustworthy","slug":"freeeval-a-modular-framework-for-trustworthy","title":"FreeEval: A Modular Framework for Trustworthy and Efficient Evaluation of Large Language Models","date":"2024-04-09","arxiv_id":"2404.06003","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/freeeval-a-modular-framework-for-trustworthy#ran","syntology_url":"https://syntology.ai/paper/2404.06003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06003"}},"official":{"repos":["wisdomshell/freeeval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/have-you-merged-my-model-on-the-robustness-of","slug":"have-you-merged-my-model-on-the-robustness-of","title":"Have You Merged My Model? On The Robustness of Large Language Model IP Protection Methods Against Model Merging","date":"2024-04-08","arxiv_id":"2404.05188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/have-you-merged-my-model-on-the-robustness-of#ran","syntology_url":"https://syntology.ai/paper/2404.05188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05188"}},"official":{"repos":["thuccslab/mergeguard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/moma-multimodal-llm-adapter-for-fast","slug":"moma-multimodal-llm-adapter-for-fast","title":"MoMA: Multimodal LLM Adapter for Fast Personalized Image Generation","date":"2024-04-08","arxiv_id":"2404.05674","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/moma-multimodal-llm-adapter-for-fast#ran","syntology_url":"https://syntology.ai/paper/2404.05674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05674"}},"official":{"repos":["bytedance/MoMA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/squeezeattention-2d-management-of-kv-cache-in","slug":"squeezeattention-2d-management-of-kv-cache-in","title":"SqueezeAttention: 2D Management of KV-Cache in LLM Inference via Layer-wise Optimal Budget","date":"2024-04-07","arxiv_id":"2404.04793","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/squeezeattention-2d-management-of-kv-cache-in#ran","syntology_url":"https://syntology.ai/paper/2404.04793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04793"}},"official":{"repos":["hetailang/squeezeattention"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pairaug-what-can-augmented-image-text-pairs","slug":"pairaug-what-can-augmented-image-text-pairs","title":"PairAug: What Can Augmented Image-Text Pairs Do for Radiology?","date":"2024-04-07","arxiv_id":"2404.04960","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pairaug-what-can-augmented-image-text-pairs#ran","syntology_url":"https://syntology.ai/paper/2404.04960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04960"}},"official":{"repos":["ytongxie/pairaug"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/physics-event-classification-using-large","slug":"physics-event-classification-using-large","title":"Physics Event Classification Using Large Language Models","date":"2024-04-05","arxiv_id":"2404.05752","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/physics-event-classification-using-large#ran","syntology_url":"https://syntology.ai/paper/2404.05752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05752"}},"official":{"repos":["ai4eic/ai4eichackathon2023-streamlit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt4-video-advancing-multimodal-llms-for","slug":"minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","arxiv_id":"2404.03413","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt4-video-advancing-multimodal-llms-for#ran","syntology_url":"https://syntology.ai/paper/2404.03413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03413"}},"official":{"repos":["Vision-CAIR/MiniGPT4-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autowebglm-bootstrap-and-reinforce-a-large","slug":"autowebglm-bootstrap-and-reinforce-a-large","title":"AutoWebGLM: A Large Language Model-based Web Navigating Agent","date":"2024-04-04","arxiv_id":"2404.03648","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/autowebglm-bootstrap-and-reinforce-a-large#ran","syntology_url":"https://syntology.ai/paper/2404.03648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03648"}},"official":{"repos":["thudm/autowebglm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-autoregressive-modeling-scalable-image","slug":"visual-autoregressive-modeling-scalable-image","title":"Visual Autoregressive Modeling: Scalable Image Generation via Next-Scale Prediction","date":"2024-04-03","arxiv_id":"2404.02905","repositories_listed":3,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/visual-autoregressive-modeling-scalable-image#ran","syntology_url":"https://syntology.ai/paper/2404.02905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02905"}},"official":{"repos":["FoundationVision/VAR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/self-organized-agents-a-llm-multi-agent","slug":"self-organized-agents-a-llm-multi-agent","title":"Self-Organized Agents: A LLM Multi-Agent Framework toward Ultra Large-Scale Code Generation and Optimization","date":"2024-04-02","arxiv_id":"2404.02183","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-organized-agents-a-llm-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2404.02183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02183"}},"official":{"repos":["tsukushiai/self-organized-agent"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-correction-efficient-tuning-task","slug":"learning-by-correction-efficient-tuning-task","title":"Learning by Correction: Efficient Tuning Task for Zero-Shot Generative Vision-Language Reasoning","date":"2024-04-01","arxiv_id":"2404.00909","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-correction-efficient-tuning-task#ran","syntology_url":"https://syntology.ai/paper/2404.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00909"}},"official":{"repos":["shtuplus/iccc_cvpr2024"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/query-performance-prediction-using-relevance","slug":"query-performance-prediction-using-relevance","title":"Query Performance Prediction using Relevance Judgments Generated by Large Language Models","date":"2024-04-01","arxiv_id":"2404.01012","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/query-performance-prediction-using-relevance#ran","syntology_url":"https://syntology.ai/paper/2404.01012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01012"}},"official":{"repos":["chuanmeng/qpp-genre"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-preference-optimization-of-video-large","slug":"direct-preference-optimization-of-video-large","title":"Direct Preference Optimization of Video Large Multimodal Models from Language Model Reward","date":"2024-04-01","arxiv_id":"2404.01258","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-preference-optimization-of-video-large#ran","syntology_url":"https://syntology.ai/paper/2404.01258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01258"}},"official":{"repos":["riflezhang/llava-hound-dpo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/developing-safe-and-responsible-large","slug":"developing-safe-and-responsible-large","title":"Developing Safe and Responsible Large Language Model : Can We Balance Bias Reduction and Language Understanding in Large Language Models?","date":"2024-04-01","arxiv_id":"2404.01399","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/developing-safe-and-responsible-large#ran","syntology_url":"https://syntology.ai/paper/2404.01399","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01399"}},"official":{"repos":["shainarazavi/safe-responsible-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/m3d-advancing-3d-medical-image-analysis-with","slug":"m3d-advancing-3d-medical-image-analysis-with","title":"M3D: Advancing 3D Medical Image Analysis with Multi-Modal Large Language Models","date":"2024-03-31","arxiv_id":"2404.00578","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m3d-advancing-3d-medical-image-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2404.00578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00578"}},"official":{"repos":["baai-dcai/m3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wavllm-towards-robust-and-adaptive-speech","slug":"wavllm-towards-robust-and-adaptive-speech","title":"WavLLM: Towards Robust and Adaptive Speech Large Language Model","date":"2024-03-31","arxiv_id":"2404.00656","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wavllm-towards-robust-and-adaptive-speech#ran","syntology_url":"https://syntology.ai/paper/2404.00656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00656"}},"official":{"repos":["microsoft/speecht5"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-vision-language-models-understand-compound","slug":"do-vision-language-models-understand-compound","title":"Do Vision-Language Models Understand Compound Nouns?","date":"2024-03-30","arxiv_id":"2404.00419","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/do-vision-language-models-understand-compound#ran","syntology_url":"https://syntology.ai/paper/2404.00419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00419"}},"official":{"repos":["sonalkum/compun"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mango-a-benchmark-for-evaluating-mapping-and","slug":"mango-a-benchmark-for-evaluating-mapping-and","title":"MANGO: A Benchmark for Evaluating Mapping and Navigation Abilities of Large Language Models","date":"2024-03-29","arxiv_id":"2403.19913","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mango-a-benchmark-for-evaluating-mapping-and#ran","syntology_url":"https://syntology.ai/paper/2403.19913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19913"}},"official":{"repos":["oaklight/mango"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tablellm-enabling-tabular-data-manipulation","slug":"tablellm-enabling-tabular-data-manipulation","title":"TableLLM: Enabling Tabular Data Manipulation by LLMs in Real Office Usage Scenarios","date":"2024-03-28","arxiv_id":"2403.19318","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tablellm-enabling-tabular-data-manipulation#ran","syntology_url":"https://syntology.ai/paper/2403.19318","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19318"}},"official":{"repos":["TableLLM/TableLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-recommendation-with-latent","slug":"sequential-recommendation-with-latent","title":"Sequential Recommendation with Latent Relations based on Large Language Model","date":"2024-03-27","arxiv_id":"2403.18348","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sequential-recommendation-with-latent#ran","syntology_url":"https://syntology.ai/paper/2403.18348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18348"}},"official":{"repos":["ysh-1998/lrd"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhanced-generative-recommendation-via","slug":"enhanced-generative-recommendation-via","title":"Content-Based Collaborative Generation for Recommender Systems","date":"2024-03-27","arxiv_id":"2403.18480","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhanced-generative-recommendation-via#ran","syntology_url":"https://syntology.ai/paper/2403.18480","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18480"}},"official":{"repos":["junewang0614/colarec"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optimization-based-prompt-injection-attack-to","slug":"optimization-based-prompt-injection-attack-to","title":"Optimization-based Prompt Injection Attack to LLM-as-a-Judge","date":"2024-03-26","arxiv_id":"2403.17710","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimization-based-prompt-injection-attack-to#ran","syntology_url":"https://syntology.ai/paper/2403.17710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17710"}},"official":{"repos":["shijiawenwen/judgedeceiver"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-foundation-model-utilizing-chest-ct-volumes","slug":"a-foundation-model-utilizing-chest-ct-volumes","title":"Developing Generalist Foundation Models from a Multimodal Dataset for 3D Computed Tomography","date":"2024-03-26","arxiv_id":"2403.17834","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":4,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 4 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-foundation-model-utilizing-chest-ct-volumes#ran","syntology_url":"https://syntology.ai/paper/2403.17834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17834"}},"official":{"repos":["ibrahimethemhamamci/ct-clip"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/if-clip-could-talk-understanding-vision","slug":"if-clip-could-talk-understanding-vision","title":"If CLIP Could Talk: Understanding Vision-Language Model Representations Through Their Preferred Concept Descriptions","date":"2024-03-25","arxiv_id":"2403.16442","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/if-clip-could-talk-understanding-vision#ran","syntology_url":"https://syntology.ai/paper/2403.16442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16442"}},"official":{"repos":["batsresearch/ex2"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-lingual-contextualized-phrase-retrieval","slug":"cross-lingual-contextualized-phrase-retrieval","title":"Cross-lingual Contextualized Phrase Retrieval","date":"2024-03-25","arxiv_id":"2403.16820","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cross-lingual-contextualized-phrase-retrieval#ran","syntology_url":"https://syntology.ai/paper/2403.16820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16820"}},"official":{"repos":["ghrua/ccpr_release"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-with-human-judgement-the-role-of","slug":"aligning-with-human-judgement-the-role-of","title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","date":"2024-03-25","arxiv_id":"2403.16950","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-with-human-judgement-the-role-of#ran","syntology_url":"https://syntology.ai/paper/2403.16950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16950"}},"official":{"repos":["cambridgeltl/pairs"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dreamlip-language-image-pre-training-with","slug":"dreamlip-language-image-pre-training-with","title":"DreamLIP: Language-Image Pre-training with Long Captions","date":"2024-03-25","arxiv_id":"2403.17007","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dreamlip-language-image-pre-training-with#ran","syntology_url":"https://syntology.ai/paper/2403.17007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17007"}},"official":{"repos":["zyf0619sjtu/DreamLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/repairagent-an-autonomous-llm-based-agent-for","slug":"repairagent-an-autonomous-llm-based-agent-for","title":"RepairAgent: An Autonomous, LLM-Based Agent for Program Repair","date":"2024-03-25","arxiv_id":"2403.17134","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/repairagent-an-autonomous-llm-based-agent-for#ran","syntology_url":"https://syntology.ai/paper/2403.17134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17134"}},"official":{"repos":["sola-st/RepairAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-prumerge-adaptive-token-reduction-for","slug":"llava-prumerge-adaptive-token-reduction-for","title":"LLaVA-PruMerge: Adaptive Token Reduction for Efficient Large Multimodal Models","date":"2024-03-22","arxiv_id":"2403.15388","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-prumerge-adaptive-token-reduction-for#ran","syntology_url":"https://syntology.ai/paper/2403.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15388"}},"official":null}},{"url":"/paper/wikifactdiff-a-large-realistic-and-temporally","slug":"wikifactdiff-a-large-realistic-and-temporally","title":"WikiFactDiff: A Large, Realistic, and Temporally Adaptable Dataset for Atomic Factual Knowledge Update in Causal Language Models","date":"2024-03-21","arxiv_id":"2403.14364","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/wikifactdiff-a-large-realistic-and-temporally#ran","syntology_url":"https://syntology.ai/paper/2403.14364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14364"}},"official":{"repos":["orange-opensource/wikifactdiff"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/cobra-extending-mamba-to-multi-modal-large","slug":"cobra-extending-mamba-to-multi-modal-large","title":"Cobra: Extending Mamba to Multi-Modal Large Language Model for Efficient Inference","date":"2024-03-21","arxiv_id":"2403.14520","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cobra-extending-mamba-to-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2403.14520","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14520"}},"official":{"repos":["h-zhao1997/cobra"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-large-language-model-enhanced-sequential","slug":"a-large-language-model-enhanced-sequential","title":"A Large Language Model Enhanced Sequential Recommender for Joint Video and Comment Recommendation","date":"2024-03-20","arxiv_id":"2403.13574","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-large-language-model-enhanced-sequential#ran","syntology_url":"https://syntology.ai/paper/2403.13574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13574"}},"official":{"repos":["rucaibox/lsvcr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-3-large-language-model-based-task-and","slug":"llm-3-large-language-model-based-task-and","title":"LLM3:Large Language Model-based Task and Motion Planning with Motion Failure Reasoning","date":"2024-03-18","arxiv_id":"2403.11552","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-3-large-language-model-based-task-and#ran","syntology_url":"https://syntology.ai/paper/2403.11552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11552"}},"official":{"repos":["assassinws/llm-tamp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-prompting-for-automating-zero-shot","slug":"meta-prompting-for-automating-zero-shot","title":"Meta-Prompting for Automating Zero-shot Visual Recognition with LLMs","date":"2024-03-18","arxiv_id":"2403.11755","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meta-prompting-for-automating-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2403.11755","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11755"}},"official":{"repos":["jmiemirza/meta-prompting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/subjective-aligned-dateset-and-metric-for","slug":"subjective-aligned-dateset-and-metric-for","title":"Subjective-Aligned Dataset and Metric for Text-to-Video Quality Assessment","date":"2024-03-18","arxiv_id":"2403.11956","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/subjective-aligned-dateset-and-metric-for#ran","syntology_url":"https://syntology.ai/paper/2403.11956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11956"}},"official":{"repos":["qmme/t2vqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selfie-self-interpretation-of-large-language","slug":"selfie-self-interpretation-of-large-language","title":"SelfIE: Self-Interpretation of Large Language Model Embeddings","date":"2024-03-16","arxiv_id":"2403.10949","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/selfie-self-interpretation-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2403.10949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.10949"}},"official":{"repos":["tonychenxyz/selfie"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-medical-multi-modal-contrastive","slug":"improving-medical-multi-modal-contrastive","title":"Improving Medical Multi-modal Contrastive Learning with Expert Annotations","date":"2024-03-15","arxiv_id":"2403.10153","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":12,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/improving-medical-multi-modal-contrastive#ran","syntology_url":"https://syntology.ai/paper/2403.10153","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.10153"}},"official":{"repos":["ykumards/eclip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/what-was-your-prompt-a-remote-keylogging","slug":"what-was-your-prompt-a-remote-keylogging","title":"What Was Your Prompt? A Remote Keylogging Attack on AI Assistants","date":"2024-03-14","arxiv_id":"2403.09751","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-was-your-prompt-a-remote-keylogging#ran","syntology_url":"https://syntology.ai/paper/2403.09751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09751"}},"official":{"repos":["royweiss1/GPT_Keylogger"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/emergence-of-social-norms-in-large-language","slug":"emergence-of-social-norms-in-large-language","title":"Emergence of Social Norms in Generative Agent Societies: Principles and Architecture","date":"2024-03-13","arxiv_id":"2403.08251","repositories_listed":1,"syntology":{"n":23,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":12,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":23,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/emergence-of-social-norms-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2403.08251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08251"}},"official":{"repos":["sxswz213/crsec"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/coin-a-benchmark-of-continual-instruction","slug":"coin-a-benchmark-of-continual-instruction","title":"CoIN: A Benchmark of Continual Instruction tuNing for Multimodel Large Language Model","date":"2024-03-13","arxiv_id":"2403.08350","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coin-a-benchmark-of-continual-instruction#ran","syntology_url":"https://syntology.ai/paper/2403.08350","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08350"}},"official":{"repos":["zackschen/coin"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sotopia-p-interactive-learning-of-socially","slug":"sotopia-p-interactive-learning-of-socially","title":"SOTOPIA-$π$: Interactive Learning of Socially Intelligent Language Agents","date":"2024-03-13","arxiv_id":"2403.08715","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sotopia-p-interactive-learning-of-socially#ran","syntology_url":"https://syntology.ai/paper/2403.08715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.08715"}},"official":{"repos":["sotopia-lab/sotopia-pi"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/svd-llm-truncation-aware-singular-value","slug":"svd-llm-truncation-aware-singular-value","title":"SVD-LLM: Truncation-aware Singular Value Decomposition for Large Language Model Compression","date":"2024-03-12","arxiv_id":"2403.07378","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/svd-llm-truncation-aware-singular-value#ran","syntology_url":"https://syntology.ai/paper/2403.07378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07378"}},"official":{"repos":["aiot-mlsys-lab/svd-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/decomposing-disease-descriptions-for-enhanced","slug":"decomposing-disease-descriptions-for-enhanced","title":"Decomposing Disease Descriptions for Enhanced Pathology Detection: A Multi-Aspect Vision-Language Pre-training Framework","date":"2024-03-12","arxiv_id":"2403.07636","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/decomposing-disease-descriptions-for-enhanced#ran","syntology_url":"https://syntology.ai/paper/2403.07636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07636"}},"official":{"repos":["hieuphan33/mavl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-text-frozen-large-language-models-in","slug":"beyond-text-frozen-large-language-models-in","title":"Beyond Text: Frozen Large Language Models in Visual Signal Comprehension","date":"2024-03-12","arxiv_id":"2403.07874","repositories_listed":1,"syntology":{"n":26,"n_ran":17,"n_constructed":0,"n_ran_checked":6,"n_instrument":11,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":26,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 11 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-text-frozen-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2403.07874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07874"}},"official":{"repos":["zh460045050/v2l-tokenizer"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/drivedreamer-2-llm-enhanced-world-models-for","slug":"drivedreamer-2-llm-enhanced-world-models-for","title":"DriveDreamer-2: LLM-Enhanced World Models for Diverse Driving Video Generation","date":"2024-03-11","arxiv_id":"2403.06845","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drivedreamer-2-llm-enhanced-world-models-for#ran","syntology_url":"https://syntology.ai/paper/2403.06845","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.06845"}},"official":null}},{"url":"/paper/monitoring-ai-modified-content-at-scale-a","slug":"monitoring-ai-modified-content-at-scale-a","title":"Monitoring AI-Modified Content at Scale: A Case Study on the Impact of ChatGPT on AI Conference Peer Reviews","date":"2024-03-11","arxiv_id":"2403.07183","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/monitoring-ai-modified-content-at-scale-a#ran","syntology_url":"https://syntology.ai/paper/2403.07183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07183"}},"official":{"repos":["Weixin-Liang/Mapping-the-Increasing-Use-of-LLMs-in-Scientific-Papers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ella-equip-diffusion-models-with-llm-for","slug":"ella-equip-diffusion-models-with-llm-for","title":"ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment","date":"2024-03-08","arxiv_id":"2403.05135","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ella-equip-diffusion-models-with-llm-for#ran","syntology_url":"https://syntology.ai/paper/2403.05135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05135"}},"official":null}},{"url":"/paper/tapilot-crossing-benchmarking-and-evolving","slug":"tapilot-crossing-benchmarking-and-evolving","title":"Tapilot-Crossing: Benchmarking and Evolving LLMs Towards Interactive Data Analysis Agents","date":"2024-03-08","arxiv_id":"2403.05307","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tapilot-crossing-benchmarking-and-evolving#ran","syntology_url":"https://syntology.ai/paper/2403.05307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05307"}},"official":{"repos":["tapilot-crossing/tapilot_code"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cat-enhancing-multimodal-large-language-model","slug":"cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","arxiv_id":"2403.04640","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cat-enhancing-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.04640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04640"}},"official":{"repos":["rikeilong/bay-cat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/injecagent-benchmarking-indirect-prompt","slug":"injecagent-benchmarking-indirect-prompt","title":"InjecAgent: Benchmarking Indirect Prompt Injections in Tool-Integrated Large Language Model Agents","date":"2024-03-05","arxiv_id":"2403.02691","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/injecagent-benchmarking-indirect-prompt#ran","syntology_url":"https://syntology.ai/paper/2403.02691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02691"}},"official":{"repos":["uiuc-kang-lab/injecagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-walk-debiasing-multi-hop-fact","slug":"causal-walk-debiasing-multi-hop-fact","title":"Causal Walk: Debiasing Multi-Hop Fact Verification with Front-Door Adjustment","date":"2024-03-05","arxiv_id":"2403.02698","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/causal-walk-debiasing-multi-hop-fact#ran","syntology_url":"https://syntology.ai/paper/2403.02698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02698"}},"official":{"repos":["zcccccz/causalwalk"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/android-in-the-zoo-chain-of-action-thought","slug":"android-in-the-zoo-chain-of-action-thought","title":"Android in the Zoo: Chain-of-Action-Thought for GUI Agents","date":"2024-03-05","arxiv_id":"2403.02713","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/android-in-the-zoo-chain-of-action-thought#ran","syntology_url":"https://syntology.ai/paper/2403.02713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02713"}},"official":{"repos":["imnearth/coat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-llm-as-a-judge-for-llm","slug":"an-empirical-study-of-llm-as-a-judge-for-llm","title":"An Empirical Study of LLM-as-a-Judge for LLM Evaluation: Fine-tuned Judge Model is not a General Substitute for GPT-4","date":"2024-03-05","arxiv_id":"2403.02839","repositories_listed":1,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":18,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-empirical-study-of-llm-as-a-judge-for-llm#ran","syntology_url":"https://syntology.ai/paper/2403.02839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02839"}},"official":{"repos":["huihuichyan/unlimitedjudge"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-instruction-tuned-llms-with-fine","slug":"multi-modal-instruction-tuned-llms-with-fine","title":"Multi-modal Instruction Tuned LLMs with Fine-grained Visual Perception","date":"2024-03-05","arxiv_id":"2403.02969","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-modal-instruction-tuned-llms-with-fine#ran","syntology_url":"https://syntology.ai/paper/2403.02969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02969"}},"official":{"repos":["jwh97nn/anyref"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/guardt2i-defending-text-to-image-models-from","slug":"guardt2i-defending-text-to-image-models-from","title":"GuardT2I: Defending Text-to-Image Models from Adversarial Prompts","date":"2024-03-03","arxiv_id":"2403.01446","repositories_listed":3,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":4,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/guardt2i-defending-text-to-image-models-from#ran","syntology_url":"https://syntology.ai/paper/2403.01446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01446"}},"official":{"repos":["cure-lab/guardt2i"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/lab-large-scale-alignment-for-chatbots","slug":"lab-large-scale-alignment-for-chatbots","title":"LAB: Large-Scale Alignment for ChatBots","date":"2024-03-02","arxiv_id":"2403.01081","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lab-large-scale-alignment-for-chatbots#ran","syntology_url":"https://syntology.ai/paper/2403.01081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01081"}},"official":null}},{"url":"/paper/opengraph-towards-open-graph-foundation","slug":"opengraph-towards-open-graph-foundation","title":"OpenGraph: Towards Open Graph Foundation Models","date":"2024-03-02","arxiv_id":"2403.01121","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/opengraph-towards-open-graph-foundation#ran","syntology_url":"https://syntology.ai/paper/2403.01121","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01121"}},"official":{"repos":["hkuds/opengraph"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/intactkv-improving-large-language-model","slug":"intactkv-improving-large-language-model","title":"IntactKV: Improving Large Language Model Quantization by Keeping Pivot Tokens Intact","date":"2024-03-02","arxiv_id":"2403.01241","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/intactkv-improving-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.01241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01241"}},"official":{"repos":["ruikangliu/IntactKV"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/flexllm-a-system-for-co-serving-large","slug":"flexllm-a-system-for-co-serving-large","title":"FlexLLM: A System for Co-Serving Large Language Model Inference and Parameter-Efficient Finetuning","date":"2024-02-29","arxiv_id":"2402.18789","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/flexllm-a-system-for-co-serving-large#ran","syntology_url":"https://syntology.ai/paper/2402.18789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18789"}},"official":{"repos":["flexflow/flexflow"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-and-reducing-catastrophic","slug":"analyzing-and-reducing-catastrophic","title":"Analyzing and Reducing Catastrophic Forgetting in Parameter Efficient Tuning","date":"2024-02-29","arxiv_id":"2402.18865","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/analyzing-and-reducing-catastrophic#ran","syntology_url":"https://syntology.ai/paper/2402.18865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18865"}},"official":{"repos":["which47/llmcl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizable-whole-slide-image","slug":"generalizable-whole-slide-image","title":"Generalizable Whole Slide Image Classification with Fine-Grained Visual-Semantic Interaction","date":"2024-02-29","arxiv_id":"2402.19326","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":5,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generalizable-whole-slide-image#ran","syntology_url":"https://syntology.ai/paper/2402.19326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19326"}},"official":{"repos":["ls1rius/wsi_five"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-long-term-recommendation-with-bi","slug":"enhancing-long-term-recommendation-with-bi","title":"Large Language Models are Learnable Planners for Long-Term Recommendation","date":"2024-02-29","arxiv_id":"2403.00843","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-long-term-recommendation-with-bi#ran","syntology_url":"https://syntology.ai/paper/2403.00843","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00843"}},"official":{"repos":["jizhi-zhang/billp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-fact-assessing-multilingual-llms-multi","slug":"multi-fact-assessing-multilingual-llms-multi","title":"Multi-FAct: Assessing Factuality of Multilingual LLMs using FActScore","date":"2024-02-28","arxiv_id":"2402.18045","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-fact-assessing-multilingual-llms-multi#ran","syntology_url":"https://syntology.ai/paper/2402.18045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18045"}},"official":{"repos":["sheikhshafayat/multi-fact"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/characterizing-truthfulness-in-large-language","slug":"characterizing-truthfulness-in-large-language","title":"Characterizing Truthfulness in Large Language Model Generations with Local Intrinsic Dimension","date":"2024-02-28","arxiv_id":"2402.18048","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/characterizing-truthfulness-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.18048","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18048"}},"official":{"repos":["fanyin3639/lid-hallucinationdetection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cogbench-a-large-language-model-walks-into-a","slug":"cogbench-a-large-language-model-walks-into-a","title":"CogBench: a large language model walks into a psychology lab","date":"2024-02-28","arxiv_id":"2402.18225","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cogbench-a-large-language-model-walks-into-a#ran","syntology_url":"https://syntology.ai/paper/2402.18225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18225"}},"official":{"repos":["juliancodaforno/cogbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-language-models-for-visual-entity","slug":"grounding-language-models-for-visual-entity","title":"Grounding Language Models for Visual Entity Recognition","date":"2024-02-28","arxiv_id":"2402.18695","repositories_listed":1,"syntology":{"n":26,"n_ran":12,"n_constructed":2,"n_ran_checked":7,"n_instrument":5,"n_unverified":14,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"12 ran (of which 2 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/grounding-language-models-for-visual-entity#ran","syntology_url":"https://syntology.ai/paper/2402.18695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18695"}},"official":{"repos":["mrzilinxiao/autover"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":2,"n_ran_no_instrument_failure":7,"n_unverified":14,"ran_from_kinds":["official"]}}},{"url":"/paper/songcomposer-a-large-language-model-for-lyric","slug":"songcomposer-a-large-language-model-for-lyric","title":"SongComposer: A Large Language Model for Lyric and Melody Generation in Song Composition","date":"2024-02-27","arxiv_id":"2402.17645","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/songcomposer-a-large-language-model-for-lyric#ran","syntology_url":"https://syntology.ai/paper/2402.17645","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17645"}},"official":{"repos":["pjlab-songcomposer/songcomposer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tower-an-open-multilingual-large-language","slug":"tower-an-open-multilingual-large-language","title":"Tower: An Open Multilingual Large Language Model for Translation-Related Tasks","date":"2024-02-27","arxiv_id":"2402.17733","repositories_listed":4,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tower-an-open-multilingual-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.17733","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17733"}},"official":{"repos":["deep-spin/tower-eval","epfllm/megatron-llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shapellm-universal-3d-object-understanding","slug":"shapellm-universal-3d-object-understanding","title":"ShapeLLM: Universal 3D Object Understanding for Embodied Interaction","date":"2024-02-27","arxiv_id":"2402.17766","repositories_listed":3,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":8,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/shapellm-universal-3d-object-understanding#ran","syntology_url":"https://syntology.ai/paper/2402.17766","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17766"}},"official":{"repos":["qizekun/ShapeLLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/truthx-alleviating-hallucinations-by-editing","slug":"truthx-alleviating-hallucinations-by-editing","title":"TruthX: Alleviating Hallucinations by Editing Large Language Models in Truthful Space","date":"2024-02-27","arxiv_id":"2402.17811","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/truthx-alleviating-hallucinations-by-editing#ran","syntology_url":"https://syntology.ai/paper/2402.17811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17811"}},"official":{"repos":["ictnlp/truthx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prediction-powered-ranking-of-large-language","slug":"prediction-powered-ranking-of-large-language","title":"Prediction-Powered Ranking of Large Language Models","date":"2024-02-27","arxiv_id":"2402.17826","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/prediction-powered-ranking-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.17826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17826"}},"official":{"repos":["networks-learning/prediction-powered-ranking"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-inference-unveiled-survey-and-roofline","slug":"llm-inference-unveiled-survey-and-roofline","title":"LLM Inference Unveiled: Survey and Roofline Model Insights","date":"2024-02-26","arxiv_id":"2402.16363","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-inference-unveiled-survey-and-roofline#ran","syntology_url":"https://syntology.ai/paper/2402.16363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16363"}},"official":{"repos":["hahnyuan/llm-viewer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/repoagent-an-llm-powered-open-source","slug":"repoagent-an-llm-powered-open-source","title":"RepoAgent: An LLM-Powered Open-Source Framework for Repository-level Code Documentation Generation","date":"2024-02-26","arxiv_id":"2402.16667","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/repoagent-an-llm-powered-open-source#ran","syntology_url":"https://syntology.ai/paper/2402.16667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16667"}},"official":{"repos":["openbmb/repoagent"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/ldb-a-large-language-model-debugger-via","slug":"ldb-a-large-language-model-debugger-via","title":"Debug like a Human: A Large Language Model Debugger via Verifying Runtime Execution Step-by-step","date":"2024-02-25","arxiv_id":"2402.16906","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ldb-a-large-language-model-debugger-via#ran","syntology_url":"https://syntology.ai/paper/2402.16906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16906"}},"official":{"repos":["floridsleeves/llmdebugger"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/empowering-large-language-model-agents","slug":"empowering-large-language-model-agents","title":"Empowering Large Language Model Agents through Action Learning","date":"2024-02-24","arxiv_id":"2402.15809","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":17,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/empowering-large-language-model-agents#ran","syntology_url":"https://syntology.ai/paper/2402.15809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15809"}},"official":{"repos":["zhao-ht/learnact"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/item-side-fairness-of-large-language-model","slug":"item-side-fairness-of-large-language-model","title":"Item-side Fairness of Large Language Model-based Recommendation System","date":"2024-02-23","arxiv_id":"2402.15215","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/item-side-fairness-of-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2402.15215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15215"}},"official":{"repos":["jiangm-c/ifairlrs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/megascale-scaling-large-language-model","slug":"megascale-scaling-large-language-model","title":"MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs","date":"2024-02-23","arxiv_id":"2402.15627","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/megascale-scaling-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2402.15627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15627"}},"official":{"repos":["volcengine/vescale"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-retrieval-building-an-information","slug":"self-retrieval-building-an-information","title":"Self-Retrieval: End-to-End Information Retrieval with One Large Language Model","date":"2024-02-23","arxiv_id":"2403.00801","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/self-retrieval-building-an-information#ran","syntology_url":"https://syntology.ai/paper/2403.00801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00801"}},"official":{"repos":["icip-cas/selfretrieval","tangqiaoyu/selfretrieval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/subobject-level-image-tokenization","slug":"subobject-level-image-tokenization","title":"Subobject-level Image Tokenization","date":"2024-02-22","arxiv_id":"2402.14327","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/subobject-level-image-tokenization#ran","syntology_url":"https://syntology.ai/paper/2402.14327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14327"}},"official":{"repos":["chendelong1999/subobjects"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/relayattention-for-efficient-large-language","slug":"relayattention-for-efficient-large-language","title":"RelayAttention for Efficient Large Language Model Serving with Long System Prompts","date":"2024-02-22","arxiv_id":"2402.14808","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/relayattention-for-efficient-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.14808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14808"}},"official":{"repos":["rayleizhu/vllm-ra"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tokenization-counts-the-impact-of","slug":"tokenization-counts-the-impact-of","title":"Tokenization counts: the impact of tokenization on arithmetic in frontier LLMs","date":"2024-02-22","arxiv_id":"2402.14903","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tokenization-counts-the-impact-of#ran","syntology_url":"https://syntology.ai/paper/2402.14903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14903"}},"official":{"repos":["aadityasingh/tokenizationcounts"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trap-targeted-random-adversarial-prompt","slug":"trap-targeted-random-adversarial-prompt","title":"TRAP: Targeted Random Adversarial Prompt Honeypot for Black-Box Identification","date":"2024-02-20","arxiv_id":"2402.12991","repositories_listed":2,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trap-targeted-random-adversarial-prompt#ran","syntology_url":"https://syntology.ai/paper/2402.12991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12991"}},"official":{"repos":["framartin/trap","parameterlab/trap"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-effects-of-language","slug":"understanding-the-effects-of-language","title":"Understanding the effects of language-specific class imbalance in multilingual fine-tuning","date":"2024-02-20","arxiv_id":"2402.13016","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-the-effects-of-language#ran","syntology_url":"https://syntology.ai/paper/2402.13016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13016"}},"official":{"repos":["idiap/class-imbalance-multilingual-ft"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/softmax-probabilities-mostly-predict-large","slug":"softmax-probabilities-mostly-predict-large","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","date":"2024-02-20","arxiv_id":"2402.13213","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/softmax-probabilities-mostly-predict-large#ran","syntology_url":"https://syntology.ai/paper/2402.13213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13213"}},"official":{"repos":["bplaut/softmax-probs-predict-llm-correctness","bplaut/llm-calibration-and-correctness-prediction"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/generation-meets-verification-accelerating","slug":"generation-meets-verification-accelerating","title":"Generation Meets Verification: Accelerating Large Language Model Inference with Smart Parallel Auto-Correct Decoding","date":"2024-02-19","arxiv_id":"2402.11809","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":4,"n_instrument":5,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generation-meets-verification-accelerating#ran","syntology_url":"https://syntology.ai/paper/2402.11809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11809"}},"official":{"repos":["cteant/space","hiyouga/llama-factory"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-large-language-model-alignment-through","slug":"direct-large-language-model-alignment-through","title":"Direct Large Language Model Alignment Through Self-Rewarding Contrastive Prompt Distillation","date":"2024-02-19","arxiv_id":"2402.11907","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/direct-large-language-model-alignment-through#ran","syntology_url":"https://syntology.ai/paper/2402.11907","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11907"}},"official":{"repos":["exlaw/dlma"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/momentor-advancing-video-large-language-model","slug":"momentor-advancing-video-large-language-model","title":"Momentor: Advancing Video Large Language Model with Fine-Grained Temporal Reasoning","date":"2024-02-18","arxiv_id":"2402.11435","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":1,"n_ran_checked":2,"n_instrument":5,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/momentor-advancing-video-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2402.11435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11435"}},"official":{"repos":["dcdmllm/momentor"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-knowledge-boundary-for-large","slug":"benchmarking-knowledge-boundary-for-large","title":"Benchmarking Knowledge Boundary for Large Language Models: A Different Perspective on Model Evaluation","date":"2024-02-18","arxiv_id":"2402.11493","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-knowledge-boundary-for-large#ran","syntology_url":"https://syntology.ai/paper/2402.11493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11493"}},"official":{"repos":["pkulcwmzx/knowledge-boundary"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-driven-meta-structure","slug":"large-language-model-driven-meta-structure","title":"Large Language Model-driven Meta-structure Discovery in Heterogeneous Information Network","date":"2024-02-18","arxiv_id":"2402.11518","repositories_listed":1,"syntology":{"n":13,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/large-language-model-driven-meta-structure#ran","syntology_url":"https://syntology.ai/paper/2402.11518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11518"}},"official":{"repos":["linchen-65/restruct"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/preact-predicting-future-in-react-enhances","slug":"preact-predicting-future-in-react-enhances","title":"PreAct: Prediction Enhances Agent's Planning Ability","date":"2024-02-18","arxiv_id":"2402.11534","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/preact-predicting-future-in-react-enhances#ran","syntology_url":"https://syntology.ai/paper/2402.11534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11534"}},"official":{"repos":["fu-dayuan/preact"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/i-learn-better-if-you-speak-my-language","slug":"i-learn-better-if-you-speak-my-language","title":"I Learn Better If You Speak My Language: Understanding the Superior Performance of Fine-Tuning Large Language Models with LLM-Generated Responses","date":"2024-02-17","arxiv_id":"2402.11192","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/i-learn-better-if-you-speak-my-language#ran","syntology_url":"https://syntology.ai/paper/2402.11192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11192"}},"official":{"repos":["xuanren4470/i-learn-better-if-you-speak-my-language"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/collavo-crayon-large-language-and-vision","slug":"collavo-crayon-large-language-and-vision","title":"CoLLaVO: Crayon Large Language and Vision mOdel","date":"2024-02-17","arxiv_id":"2402.11248","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collavo-crayon-large-language-and-vision#ran","syntology_url":"https://syntology.ai/paper/2402.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11248"}},"official":{"repos":["ByungKwanLee/CoLLaVO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dissecting-human-and-llm-preferences","slug":"dissecting-human-and-llm-preferences","title":"Dissecting Human and LLM Preferences","date":"2024-02-17","arxiv_id":"2402.11296","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dissecting-human-and-llm-preferences#ran","syntology_url":"https://syntology.ai/paper/2402.11296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11296"}},"official":{"repos":["gair-nlp/preference-dissection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rag-driver-generalisable-driving-explanations","slug":"rag-driver-generalisable-driving-explanations","title":"RAG-Driver: Generalisable Driving Explanations with Retrieval-Augmented In-Context Learning in Multi-Modal Large Language Model","date":"2024-02-16","arxiv_id":"2402.10828","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rag-driver-generalisable-driving-explanations#ran","syntology_url":"https://syntology.ai/paper/2402.10828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10828"}},"official":null}},{"url":"/paper/vqattack-transferable-adversarial-attacks-on","slug":"vqattack-transferable-adversarial-attacks-on","title":"VQAttack: Transferable Adversarial Attacks on Visual Question Answering via Pre-trained Models","date":"2024-02-16","arxiv_id":"2402.11083","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vqattack-transferable-adversarial-attacks-on#ran","syntology_url":"https://syntology.ai/paper/2402.11083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11083"}},"official":null}},{"url":"/paper/generative-representational-instruction","slug":"generative-representational-instruction","title":"Generative Representational Instruction Tuning","date":"2024-02-15","arxiv_id":"2402.09906","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generative-representational-instruction#ran","syntology_url":"https://syntology.ai/paper/2402.09906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09906"}},"official":{"repos":["contextualai/gritlm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-augmented-in-context-learning-for","slug":"self-augmented-in-context-learning-for","title":"Self-Augmented In-Context Learning for Unsupervised Word Translation","date":"2024-02-15","arxiv_id":"2402.10024","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/self-augmented-in-context-learning-for#ran","syntology_url":"https://syntology.ai/paper/2402.10024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10024"}},"official":{"repos":["cambridgeltl/sail-bli"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chemreasoner-heuristic-search-over-a-large","slug":"chemreasoner-heuristic-search-over-a-large","title":"ChemReasoner: Heuristic Search over a Large Language Model's Knowledge Space using Quantum-Chemical Feedback","date":"2024-02-15","arxiv_id":"2402.10980","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chemreasoner-heuristic-search-over-a-large#ran","syntology_url":"https://syntology.ai/paper/2402.10980","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10980"}},"official":{"repos":["pnnl/chemreasoner"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rapid-adoption-hidden-risks-the-dual-impact","slug":"rapid-adoption-hidden-risks-the-dual-impact","title":"Instruction Backdoor Attacks Against Customized LLMs","date":"2024-02-14","arxiv_id":"2402.09179","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rapid-adoption-hidden-risks-the-dual-impact#ran","syntology_url":"https://syntology.ai/paper/2402.09179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09179"}},"official":{"repos":["zhangrui4041/instruction_backdoor_attack"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"e779f8f496817875c4647c5ceabe657db9831533d1185612be8d9bcf40dad00c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}