{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/ran/3","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":8,"rows_per_page":100,"rows":[201,300],"of":749,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking/papers/ran/1","prev":"/task/benchmarking/papers/ran/2","next":"/task/benchmarking/papers/ran/4","papers":[{"url":"/paper/benchmarking-structure-based-three","slug":"benchmarking-structure-based-three","title":"Benchmarking structure-based three-dimensional molecular generative models using GenBench3D: ligand conformation quality matters","date":"2024-07-05","arxiv_id":"2407.04424","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-structure-based-three#ran","syntology_url":"https://syntology.ai/paper/2407.04424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04424"}},"official":{"repos":["bbaillif/genbench3d"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/craftium-an-extensible-framework-for-creating","slug":"craftium-an-extensible-framework-for-creating","title":"Craftium: An Extensible Framework for Creating Reinforcement Learning Environments","date":"2024-07-04","arxiv_id":"2407.03969","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/craftium-an-extensible-framework-for-creating#ran","syntology_url":"https://syntology.ai/paper/2407.03969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03969"}},"official":{"repos":["mikelma/craftium"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-complex-instruction-following","slug":"benchmarking-complex-instruction-following","title":"Benchmarking Complex Instruction-Following with Multiple Constraints Composition","date":"2024-07-04","arxiv_id":"2407.03978","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-complex-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2407.03978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03978"}},"official":{"repos":["thu-coai/complexbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/emotion-and-intent-joint-understanding-in","slug":"emotion-and-intent-joint-understanding-in","title":"Emotion and Intent Joint Understanding in Multimodal Conversation: A Benchmarking Dataset","date":"2024-07-03","arxiv_id":"2407.02751","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emotion-and-intent-joint-understanding-in#ran","syntology_url":"https://syntology.ai/paper/2407.02751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02751"}},"official":{"repos":["mc-eiu/mc-eiu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coir-a-comprehensive-benchmark-for-code","slug":"coir-a-comprehensive-benchmark-for-code","title":"CoIR: A Comprehensive Benchmark for Code Information Retrieval Models","date":"2024-07-03","arxiv_id":"2407.02883","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coir-a-comprehensive-benchmark-for-code#ran","syntology_url":"https://syntology.ai/paper/2407.02883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02883"}},"official":{"repos":["coir-team/coir"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gracore-benchmarking-graph-comprehension-and","slug":"gracore-benchmarking-graph-comprehension-and","title":"GraCoRe: Benchmarking Graph Comprehension and Complex Reasoning in Large Language Models","date":"2024-07-03","arxiv_id":"2407.02936","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gracore-benchmarking-graph-comprehension-and#ran","syntology_url":"https://syntology.ai/paper/2407.02936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02936"}},"official":{"repos":["zikeyuan/gracore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/comics-datasets-framework-mix-of-comics","slug":"comics-datasets-framework-mix-of-comics","title":"Comics Datasets Framework: Mix of Comics datasets for detection benchmarking","date":"2024-07-03","arxiv_id":"2407.03540","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/comics-datasets-framework-mix-of-comics#ran","syntology_url":"https://syntology.ai/paper/2407.03540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03540"}},"official":{"repos":["emanuelevivoli/cdf"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/occlusion-aware-seamless-segmentation","slug":"occlusion-aware-seamless-segmentation","title":"Occlusion-Aware Seamless Segmentation","date":"2024-07-02","arxiv_id":"2407.02182","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/occlusion-aware-seamless-segmentation#ran","syntology_url":"https://syntology.ai/paper/2407.02182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02182"}},"official":{"repos":["yihong-97/oass"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fairmedfm-fairness-benchmarking-for-medical","slug":"fairmedfm-fairness-benchmarking-for-medical","title":"FairMedFM: Fairness Benchmarking for Medical Imaging Foundation Models","date":"2024-07-01","arxiv_id":"2407.00983","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fairmedfm-fairness-benchmarking-for-medical#ran","syntology_url":"https://syntology.ai/paper/2407.00983","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00983"}},"official":{"repos":["FairMedFM/FairMedFM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mobile-bench-an-evaluation-benchmark-for-llm","slug":"mobile-bench-an-evaluation-benchmark-for-llm","title":"Mobile-Bench: An Evaluation Benchmark for LLM-based Mobile Agents","date":"2024-07-01","arxiv_id":"2407.00993","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mobile-bench-an-evaluation-benchmark-for-llm#ran","syntology_url":"https://syntology.ai/paper/2407.00993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00993"}},"official":{"repos":["XiaoMi/MobileBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-common-flaws-in-the-evaluation-of","slug":"overcoming-common-flaws-in-the-evaluation-of","title":"Overcoming Common Flaws in the Evaluation of Selective Classification Systems","date":"2024-07-01","arxiv_id":"2407.01032","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/overcoming-common-flaws-in-the-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2407.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01032"}},"official":{"repos":["iml-dkfz/fd-shifts"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bergen-a-benchmarking-library-for-retrieval","slug":"bergen-a-benchmarking-library-for-retrieval","title":"BERGEN: A Benchmarking Library for Retrieval-Augmented Generation","date":"2024-07-01","arxiv_id":"2407.01102","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bergen-a-benchmarking-library-for-retrieval#ran","syntology_url":"https://syntology.ai/paper/2407.01102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01102"}},"official":{"repos":["naver/bergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-predictive-coding-networks-made","slug":"benchmarking-predictive-coding-networks-made","title":"Benchmarking Predictive Coding Networks -- Made Simple","date":"2024-07-01","arxiv_id":"2407.01163","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-predictive-coding-networks-made#ran","syntology_url":"https://syntology.ai/paper/2407.01163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01163"}},"official":{"repos":["liukidar/pcax"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-agents-that-matter","slug":"ai-agents-that-matter","title":"AI Agents That Matter","date":"2024-07-01","arxiv_id":"2407.01502","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai-agents-that-matter#ran","syntology_url":"https://syntology.ai/paper/2407.01502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01502"}},"official":{"repos":["benediktstroebl/agent-evals"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmlongbench-doc-benchmarking-long-context","slug":"mmlongbench-doc-benchmarking-long-context","title":"MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations","date":"2024-07-01","arxiv_id":"2407.01523","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmlongbench-doc-benchmarking-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.01523","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01523"}},"official":null}},{"url":"/paper/grapharena-benchmarking-large-language-models","slug":"grapharena-benchmarking-large-language-models","title":"GraphArena: Benchmarking Large Language Models on Graph Computational Problems","date":"2024-06-29","arxiv_id":"2407.00379","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grapharena-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.00379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00379"}},"official":{"repos":["squareroot3/grapharena"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unigen-a-unified-framework-for-textual","slug":"unigen-a-unified-framework-for-textual","title":"UniGen: A Unified Framework for Textual Dataset Generation Using Large Language Models","date":"2024-06-27","arxiv_id":"2406.18966","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unigen-a-unified-framework-for-textual#ran","syntology_url":"https://syntology.ai/paper/2406.18966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18966"}},"official":{"repos":["howiehwong/unigen"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-foundation-world-models-for","slug":"multimodal-foundation-world-models-for","title":"GenRL: Multimodal-foundation world models for generalization in embodied agents","date":"2024-06-26","arxiv_id":"2406.18043","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multimodal-foundation-world-models-for#ran","syntology_url":"https://syntology.ai/paper/2406.18043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18043"}},"official":{"repos":["mazpie/genrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mathodyssey-benchmarking-mathematical-problem","slug":"mathodyssey-benchmarking-mathematical-problem","title":"MathOdyssey: Benchmarking Mathematical Problem-Solving Skills in Large Language Models Using Odyssey Math Data","date":"2024-06-26","arxiv_id":"2406.18321","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathodyssey-benchmarking-mathematical-problem#ran","syntology_url":"https://syntology.ai/paper/2406.18321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18321"}},"official":null}},{"url":"/paper/leave-no-document-behind-benchmarking-long","slug":"leave-no-document-behind-benchmarking-long","title":"Leave No Document Behind: Benchmarking Long-Context LLMs with Extended Multi-Doc QA","date":"2024-06-25","arxiv_id":"2406.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/leave-no-document-behind-benchmarking-long#ran","syntology_url":"https://syntology.ai/paper/2406.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17419"}},"official":{"repos":["mozerwang/loong"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/varbench-robust-language-model-benchmarking","slug":"varbench-robust-language-model-benchmarking","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","date":"2024-06-25","arxiv_id":"2406.17681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/varbench-robust-language-model-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.17681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17681"}},"official":{"repos":["qbetterk/VarBench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inherent-challenges-of-post-hoc-membership","slug":"inherent-challenges-of-post-hoc-membership","title":"SoK: Membership Inference Attacks on LLMs are Rushing Nowhere (and How to Fix It)","date":"2024-06-25","arxiv_id":"2406.17975","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inherent-challenges-of-post-hoc-membership#ran","syntology_url":"https://syntology.ai/paper/2406.17975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17975"}},"official":{"repos":["computationalprivacy/mia_llms_benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/general-binding-affinity-guidance-for","slug":"general-binding-affinity-guidance-for","title":"General Binding Affinity Guidance for Diffusion Models in Structure-Based Drug Design","date":"2024-06-24","arxiv_id":"2406.16821","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/general-binding-affinity-guidance-for#ran","syntology_url":"https://syntology.ai/paper/2406.16821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16821"}},"official":{"repos":["ask-berkeley/badger-sbdd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ragnarok-a-reusable-rag-framework-and","slug":"ragnarok-a-reusable-rag-framework-and","title":"Ragnarök: A Reusable RAG Framework and Baselines for TREC 2024 Retrieval-Augmented Generation Track","date":"2024-06-24","arxiv_id":"2406.16828","repositories_listed":2,"syntology":{"n":23,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ragnarok-a-reusable-rag-framework-and#ran","syntology_url":"https://syntology.ai/paper/2406.16828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16828"}},"official":{"repos":["castorini/ragnarok"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dreambench-a-human-aligned-benchmark-for","slug":"dreambench-a-human-aligned-benchmark-for","title":"DreamBench++: A Human-Aligned Benchmark for Personalized Image Generation","date":"2024-06-24","arxiv_id":"2406.16855","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dreambench-a-human-aligned-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.16855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16855"}},"official":{"repos":["yuangpeng/dreambench_plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-open-respiratory-acoustic-foundation","slug":"towards-open-respiratory-acoustic-foundation","title":"Towards Open Respiratory Acoustic Foundation Models: Pretraining and Benchmarking","date":"2024-06-23","arxiv_id":"2406.16148","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-open-respiratory-acoustic-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.16148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16148"}},"official":{"repos":["evelyn0414/opera"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hest-1k-a-dataset-for-spatial-transcriptomics","slug":"hest-1k-a-dataset-for-spatial-transcriptomics","title":"HEST-1k: A Dataset for Spatial Transcriptomics and Histology Image Analysis","date":"2024-06-23","arxiv_id":"2406.16192","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hest-1k-a-dataset-for-spatial-transcriptomics#ran","syntology_url":"https://syntology.ai/paper/2406.16192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16192"}},"official":{"repos":["mahmoodlab/hest"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bigcodebench-benchmarking-code-generation","slug":"bigcodebench-benchmarking-code-generation","title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","date":"2024-06-22","arxiv_id":"2406.15877","repositories_listed":4,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bigcodebench-benchmarking-code-generation#ran","syntology_url":"https://syntology.ai/paper/2406.15877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15877"}},"official":{"repos":["bigcode-project/bigcodebench","bigcode-project/bigcodebench-annotation"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deciphering-the-definition-of-adversarial","slug":"deciphering-the-definition-of-adversarial","title":"Deciphering the Definition of Adversarial Robustness for post-hoc OOD Detectors","date":"2024-06-21","arxiv_id":"2406.15104","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/deciphering-the-definition-of-adversarial#ran","syntology_url":"https://syntology.ai/paper/2406.15104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15104"}},"official":{"repos":["adverml/advopenood"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/navsim-data-driven-non-reactive-autonomous","slug":"navsim-data-driven-non-reactive-autonomous","title":"NAVSIM: Data-Driven Non-Reactive Autonomous Vehicle Simulation and Benchmarking","date":"2024-06-21","arxiv_id":"2406.15349","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/navsim-data-driven-non-reactive-autonomous#ran","syntology_url":"https://syntology.ai/paper/2406.15349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15349"}},"official":{"repos":["autonomousvision/navsim"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-uncertainty-quantification","slug":"benchmarking-uncertainty-quantification","title":"Benchmarking Uncertainty Quantification Methods for Large Language Models with LM-Polygraph","date":"2024-06-21","arxiv_id":"2406.15627","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-uncertainty-quantification#ran","syntology_url":"https://syntology.ai/paper/2406.15627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15627"}},"official":{"repos":["iinemo/lm-polygraph"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/beyond-optimism-exploration-with-partially","slug":"beyond-optimism-exploration-with-partially","title":"Beyond Optimism: Exploration With Partially Observable Rewards","date":"2024-06-20","arxiv_id":"2406.13909","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":10,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/beyond-optimism-exploration-with-partially#ran","syntology_url":"https://syntology.ai/paper/2406.13909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13909"}},"official":{"repos":["AmiiThinks/mon_mdp_neurips24"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/the-elusive-pursuit-of-replicating-pate-gan","slug":"the-elusive-pursuit-of-replicating-pate-gan","title":"The Elusive Pursuit of Reproducing PATE-GAN: Benchmarking, Auditing, Debugging","date":"2024-06-20","arxiv_id":"2406.13985","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-elusive-pursuit-of-replicating-pate-gan#ran","syntology_url":"https://syntology.ai/paper/2406.13985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13985"}},"official":{"repos":["spalabucr/pategan-audit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/weather-5k-a-large-scale-global-station","slug":"weather-5k-a-large-scale-global-station","title":"How far are today's time-series models from real-world weather forecasting applications?","date":"2024-06-20","arxiv_id":"2406.14399","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":15,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 15 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/weather-5k-a-large-scale-global-station#ran","syntology_url":"https://syntology.ai/paper/2406.14399","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14399"}},"official":{"repos":["taohan10200/weather-5k"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":15,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/african-or-european-swallow-benchmarking","slug":"african-or-european-swallow-benchmarking","title":"African or European Swallow? Benchmarking Large Vision-Language Models for Fine-Grained Object Classification","date":"2024-06-20","arxiv_id":"2406.14496","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/african-or-european-swallow-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.14496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14496"}},"official":{"repos":["gregor-ge/foci-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-benchmarking-study-of-kolmogorov-arnold","slug":"a-benchmarking-study-of-kolmogorov-arnold","title":"A Benchmarking Study of Kolmogorov-Arnold Networks on Tabular Data","date":"2024-06-20","arxiv_id":"2406.14529","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-benchmarking-study-of-kolmogorov-arnold#ran","syntology_url":"https://syntology.ai/paper/2406.14529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14529"}},"official":{"repos":["eleonorapoeta/benchmarking-kan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mama-mia-a-large-scale-multi-center-breast","slug":"mama-mia-a-large-scale-multi-center-breast","title":"A large-scale multicenter breast cancer DCE-MRI benchmark dataset with expert segmentations","date":"2024-06-19","arxiv_id":"2406.13844","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mama-mia-a-large-scale-multi-center-breast#ran","syntology_url":"https://syntology.ai/paper/2406.13844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13844"}},"official":{"repos":["lidiagarrucho/mama-mia"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/webcanvas-benchmarking-web-agents-in-online","slug":"webcanvas-benchmarking-web-agents-in-online","title":"WebCanvas: Benchmarking Web Agents in Online Environments","date":"2024-06-18","arxiv_id":"2406.12373","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/webcanvas-benchmarking-web-agents-in-online#ran","syntology_url":"https://syntology.ai/paper/2406.12373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12373"}},"official":{"repos":["imeanai/webcanvas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-multi-image-understanding-in","slug":"benchmarking-multi-image-understanding-in","title":"Benchmarking Multi-Image Understanding in Vision and Language Models: Perception, Knowledge, Reasoning, and Multi-Hop Reasoning","date":"2024-06-18","arxiv_id":"2406.12742","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-multi-image-understanding-in#ran","syntology_url":"https://syntology.ai/paper/2406.12742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12742"}},"official":{"repos":["dtennant/mirb_eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/olympicarena-benchmarking-multi-discipline","slug":"olympicarena-benchmarking-multi-discipline","title":"OlympicArena: Benchmarking Multi-discipline Cognitive Reasoning for Superintelligent AI","date":"2024-06-18","arxiv_id":"2406.12753","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/olympicarena-benchmarking-multi-discipline#ran","syntology_url":"https://syntology.ai/paper/2406.12753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12753"}},"official":{"repos":["gair-nlp/olympicarena"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-needle-in-a-haystack-benchmarking","slug":"multimodal-needle-in-a-haystack-benchmarking","title":"Multimodal Needle in a Haystack: Benchmarking Long-Context Capability of Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11230","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-needle-in-a-haystack-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.11230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11230"}},"official":{"repos":["wang-ml-lab/multimodal-needle-in-a-haystack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/standardizing-structural-causal-models","slug":"standardizing-structural-causal-models","title":"Standardizing Structural Causal Models","date":"2024-06-17","arxiv_id":"2406.11601","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/standardizing-structural-causal-models#ran","syntology_url":"https://syntology.ai/paper/2406.11601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11601"}},"official":{"repos":["werkaaa/iscm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/job-sdf-a-multi-granularity-dataset-for-job","slug":"job-sdf-a-multi-granularity-dataset-for-job","title":"Job-SDF: A Multi-Granularity Dataset for Job Skill Demand Forecasting and Benchmarking","date":"2024-06-17","arxiv_id":"2406.11920","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/job-sdf-a-multi-granularity-dataset-for-job#ran","syntology_url":"https://syntology.ai/paper/2406.11920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11920"}},"official":{"repos":["job-sdf/benchmark"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rwku-benchmarking-real-world-knowledge","slug":"rwku-benchmarking-real-world-knowledge","title":"RWKU: Benchmarking Real-World Knowledge Unlearning for Large Language Models","date":"2024-06-16","arxiv_id":"2406.10890","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rwku-benchmarking-real-world-knowledge#ran","syntology_url":"https://syntology.ai/paper/2406.10890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10890"}},"official":{"repos":["jinzhuoran/rwku"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-label-noise-in-instance","slug":"benchmarking-label-noise-in-instance","title":"Benchmarking Label Noise in Instance Segmentation: Spatial Noise Matters","date":"2024-06-16","arxiv_id":"2406.10891","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-label-noise-in-instance#ran","syntology_url":"https://syntology.ai/paper/2406.10891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10891"}},"official":{"repos":["eden500/Noisy-Labels-Instance-Segmentation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/novobench-benchmarking-deep-learning-based-de","slug":"novobench-benchmarking-deep-learning-based-de","title":"NovoBench: Benchmarking Deep Learning-based De Novo Peptide Sequencing Methods in Proteomics","date":"2024-06-16","arxiv_id":"2406.11906","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/novobench-benchmarking-deep-learning-based-de#ran","syntology_url":"https://syntology.ai/paper/2406.11906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11906"}},"official":null}},{"url":"/paper/benchmarking-children-s-asr-with-supervised","slug":"benchmarking-children-s-asr-with-supervised","title":"Benchmarking Children's ASR with Supervised and Self-supervised Speech Foundation Models","date":"2024-06-15","arxiv_id":"2406.10507","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-children-s-asr-with-supervised#ran","syntology_url":"https://syntology.ai/paper/2406.10507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10507"}},"official":{"repos":["Diamondfan/SPAPL_KidsASR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tgb-2-0-a-benchmark-for-learning-on-temporal","slug":"tgb-2-0-a-benchmark-for-learning-on-temporal","title":"TGB 2.0: A Benchmark for Learning on Temporal Knowledge Graphs and Heterogeneous Graphs","date":"2024-06-14","arxiv_id":"2406.09639","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tgb-2-0-a-benchmark-for-learning-on-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.09639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09639"}},"official":{"repos":["erfanloghmani/myket-android-application-market-dataset","juliagast/tgb2","shenyanghuang/tgb"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-spectral-graph-neural-networks-a","slug":"benchmarking-spectral-graph-neural-networks-a","title":"Benchmarking Spectral Graph Neural Networks: A Comprehensive Study on Effectiveness and Efficiency","date":"2024-06-14","arxiv_id":"2406.09675","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-spectral-graph-neural-networks-a#ran","syntology_url":"https://syntology.ai/paper/2406.09675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09675"}},"official":{"repos":["gdmnl/spectral-gnn-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streambench-towards-benchmarking-continuous","slug":"streambench-towards-benchmarking-continuous","title":"StreamBench: Towards Benchmarking Continuous Improvement of Language Agents","date":"2024-06-13","arxiv_id":"2406.08747","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/streambench-towards-benchmarking-continuous#ran","syntology_url":"https://syntology.ai/paper/2406.08747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08747"}},"official":{"repos":["stream-bench/stream-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sr-caco-2-a-dataset-for-confocal-fluorescence","slug":"sr-caco-2-a-dataset-for-confocal-fluorescence","title":"SR-CACO-2: A Dataset for Confocal Fluorescence Microscopy Image Super-Resolution","date":"2024-06-13","arxiv_id":"2406.09168","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sr-caco-2-a-dataset-for-confocal-fluorescence#ran","syntology_url":"https://syntology.ai/paper/2406.09168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09168"}},"official":{"repos":["sbelharbi/sr-caco-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bag-of-tricks-benchmarking-of-jailbreak","slug":"bag-of-tricks-benchmarking-of-jailbreak","title":"Bag of Tricks: Benchmarking of Jailbreak Attacks on LLMs","date":"2024-06-13","arxiv_id":"2406.09324","repositories_listed":2,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bag-of-tricks-benchmarking-of-jailbreak#ran","syntology_url":"https://syntology.ai/paper/2406.09324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09324"}},"official":{"repos":["usail-hkust/bag_of_tricks_for_llm_jailbreaking","usail-hkust/jailtrickbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/needle-in-a-video-haystack-a-scalable","slug":"needle-in-a-video-haystack-a-scalable","title":"Needle In A Video Haystack: A Scalable Synthetic Evaluator for Video MLLMs","date":"2024-06-13","arxiv_id":"2406.09367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/needle-in-a-video-haystack-a-scalable#ran","syntology_url":"https://syntology.ai/paper/2406.09367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09367"}},"official":{"repos":["joez17/videoniah"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/drivaernet-a-large-scale-multimodal-car","slug":"drivaernet-a-large-scale-multimodal-car","title":"DrivAerNet++: A Large-Scale Multimodal Car Dataset with Computational Fluid Dynamics Simulations and Deep Learning Benchmarks","date":"2024-06-13","arxiv_id":"2406.09624","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drivaernet-a-large-scale-multimodal-car#ran","syntology_url":"https://syntology.ai/paper/2406.09624","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09624"}},"official":{"repos":["mohamedelrefaie/drivaernet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/examining-post-training-quantization-for","slug":"examining-post-training-quantization-for","title":"Examining Post-Training Quantization for Mixture-of-Experts: A Benchmark","date":"2024-06-12","arxiv_id":"2406.08155","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/examining-post-training-quantization-for#ran","syntology_url":"https://syntology.ai/paper/2406.08155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08155"}},"official":{"repos":["unites-lab/moe-quantization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/causality-for-tabular-data-synthesis-a-high","slug":"causality-for-tabular-data-synthesis-a-high","title":"Causality for Tabular Data Synthesis: A High-Order Structure Causal Benchmark Framework","date":"2024-06-12","arxiv_id":"2406.08311","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causality-for-tabular-data-synthesis-a-high#ran","syntology_url":"https://syntology.ai/paper/2406.08311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08311"}},"official":{"repos":["turuibo/cautabbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tc-bench-benchmarking-temporal","slug":"tc-bench-benchmarking-temporal","title":"TC-Bench: Benchmarking Temporal Compositionality in Text-to-Video and Image-to-Video Generation","date":"2024-06-12","arxiv_id":"2406.08656","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tc-bench-benchmarking-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.08656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08656"}},"official":{"repos":["weixi-feng/tc-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audiomarkbench-benchmarking-robustness-of","slug":"audiomarkbench-benchmarking-robustness-of","title":"AudioMarkBench: Benchmarking Robustness of Audio Watermarking","date":"2024-06-11","arxiv_id":"2406.06979","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiomarkbench-benchmarking-robustness-of#ran","syntology_url":"https://syntology.ai/paper/2406.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06979"}},"official":{"repos":["moyangkuo/audiomarkbench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/discoveryworld-a-virtual-environment-for","slug":"discoveryworld-a-virtual-environment-for","title":"DISCOVERYWORLD: A Virtual Environment for Developing and Evaluating Automated Scientific Discovery Agents","date":"2024-06-10","arxiv_id":"2406.06769","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discoveryworld-a-virtual-environment-for#ran","syntology_url":"https://syntology.ai/paper/2406.06769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06769"}},"official":{"repos":["allenai/discoveryworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/icu-sepsis-a-benchmark-mdp-built-from-real","slug":"icu-sepsis-a-benchmark-mdp-built-from-real","title":"ICU-Sepsis: A Benchmark MDP Built from Real Medical Data","date":"2024-06-09","arxiv_id":"2406.05646","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/icu-sepsis-a-benchmark-mdp-built-from-real#ran","syntology_url":"https://syntology.ai/paper/2406.05646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05646"}},"official":{"repos":["icu-sepsis/icu-sepsis"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/topobenchmarkx-a-framework-for-benchmarking","slug":"topobenchmarkx-a-framework-for-benchmarking","title":"TopoBench: A Framework for Benchmarking Topological Deep Learning","date":"2024-06-09","arxiv_id":"2406.06642","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/topobenchmarkx-a-framework-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.06642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06642"}},"official":{"repos":["geometric-intelligence/TopoBench","pyt-team/TopoBenchmarkX"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clog-benchmarking-continual-learning-of-image","slug":"clog-benchmarking-continual-learning-of-image","title":"CLoG: Benchmarking Continual Learning of Image Generation Models","date":"2024-06-07","arxiv_id":"2406.04584","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clog-benchmarking-continual-learning-of-image#ran","syntology_url":"https://syntology.ai/paper/2406.04584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04584"}},"official":{"repos":["linhaowei1/clog"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/wildbench-benchmarking-llms-with-challenging","slug":"wildbench-benchmarking-llms-with-challenging","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","date":"2024-06-07","arxiv_id":"2406.04770","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/wildbench-benchmarking-llms-with-challenging#ran","syntology_url":"https://syntology.ai/paper/2406.04770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04770"}},"official":{"repos":["allenai/wildbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bench2drive-towards-multi-ability","slug":"bench2drive-towards-multi-ability","title":"Bench2Drive: Towards Multi-Ability Benchmarking of Closed-Loop End-To-End Autonomous Driving","date":"2024-06-06","arxiv_id":"2406.03877","repositories_listed":4,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":17,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/bench2drive-towards-multi-ability#ran","syntology_url":"https://syntology.ai/paper/2406.03877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03877"}},"official":{"repos":["Thinklab-SJTU/Bench2Drive","autonomousvision/carla_garage","thinklab-sjtu/bench2drivezoo"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mlvu-a-comprehensive-benchmark-for-multi-task","slug":"mlvu-a-comprehensive-benchmark-for-multi-task","title":"MLVU: Benchmarking Multi-task Long Video Understanding","date":"2024-06-06","arxiv_id":"2406.04264","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mlvu-a-comprehensive-benchmark-for-multi-task#ran","syntology_url":"https://syntology.ai/paper/2406.04264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04264"}},"official":{"repos":["junjie99/mlvu","FlagOpen/FlagEmbedding"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/time-sensitive-knowledge-editing-through","slug":"time-sensitive-knowledge-editing-through","title":"Time Sensitive Knowledge Editing through Efficient Finetuning","date":"2024-06-06","arxiv_id":"2406.04496","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/time-sensitive-knowledge-editing-through#ran","syntology_url":"https://syntology.ai/paper/2406.04496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04496"}},"official":{"repos":["hiyouga/llama-factory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tidmad-time-series-dataset-for-discovering","slug":"tidmad-time-series-dataset-for-discovering","title":"TIDMAD: Time Series Dataset for Discovering Dark Matter with AI Denoising","date":"2024-06-05","arxiv_id":"2406.04378","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":8,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/tidmad-time-series-dataset-for-discovering#ran","syntology_url":"https://syntology.ai/paper/2406.04378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04378"}},"official":{"repos":["jessicafry/tidmad"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-into-clustering-of-unseen","slug":"an-empirical-study-into-clustering-of-unseen","title":"An Empirical Study into Clustering of Unseen Datasets with Self-Supervised Encoders","date":"2024-06-04","arxiv_id":"2406.02465","repositories_listed":3,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/an-empirical-study-into-clustering-of-unseen#ran","syntology_url":"https://syntology.ai/paper/2406.02465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02465"}},"official":{"repos":["scottclowe/zs-ssl-clustering"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/websuite-systematically-evaluating-why-web","slug":"websuite-systematically-evaluating-why-web","title":"WebSuite: Systematically Evaluating Why Web Agents Fail","date":"2024-06-01","arxiv_id":"2406.01623","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/websuite-systematically-evaluating-why-web#ran","syntology_url":"https://syntology.ai/paper/2406.01623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01623"}},"official":{"repos":["erichli1/websuite"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llmgeo-benchmarking-large-language-models-on","slug":"llmgeo-benchmarking-large-language-models-on","title":"LLMGeo: Benchmarking Large Language Models on Image Geolocation In-the-wild","date":"2024-05-30","arxiv_id":"2405.20363","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":15,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/llmgeo-benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2405.20363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20363"}},"official":{"repos":["yeyimilk/llmgeo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/quantitative-certification-of-bias-in-large","slug":"quantitative-certification-of-bias-in-large","title":"Quantitative Certification of Bias in Large Language Models","date":"2024-05-29","arxiv_id":"2405.18780","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/quantitative-certification-of-bias-in-large#ran","syntology_url":"https://syntology.ai/paper/2405.18780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18780"}},"official":{"repos":["uiuc-focal-lab/quacer-b"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/benchmarking-and-improving-detail-image","slug":"benchmarking-and-improving-detail-image","title":"Benchmarking and Improving Detail Image Caption","date":"2024-05-29","arxiv_id":"2405.19092","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-improving-detail-image#ran","syntology_url":"https://syntology.ai/paper/2405.19092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19092"}},"official":{"repos":["foundation-multimodal-models/capture"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathchat-benchmarking-mathematical-reasoning","slug":"mathchat-benchmarking-mathematical-reasoning","title":"MathChat: Benchmarking Mathematical Reasoning and Instruction Following in Multi-Turn Interactions","date":"2024-05-29","arxiv_id":"2405.19444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathchat-benchmarking-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2405.19444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19444"}},"official":{"repos":["zhenwen-nlp/mathchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-skeleton-based-motion-encoder","slug":"benchmarking-skeleton-based-motion-encoder","title":"Benchmarking Skeleton-based Motion Encoder Models for Clinical Applications: Estimating Parkinson's Disease Severity in Walking Sequences","date":"2024-05-28","arxiv_id":"2405.17817","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-skeleton-based-motion-encoder#ran","syntology_url":"https://syntology.ai/paper/2405.17817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17817"}},"official":{"repos":["taatiteam/motionencoders_parkinsonism_benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-improving-bird-s-eye-view","slug":"benchmarking-and-improving-bird-s-eye-view","title":"Benchmarking and Improving Bird's Eye View Perception Robustness in Autonomous Driving","date":"2024-05-27","arxiv_id":"2405.17426","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-and-improving-bird-s-eye-view#ran","syntology_url":"https://syntology.ai/paper/2405.17426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17426"}},"official":{"repos":["Daniel-xsy/RoboBEV"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-xs-low-rank-adaptation-with-extremely","slug":"lora-xs-low-rank-adaptation-with-extremely","title":"LoRA-XS: Low-Rank Adaptation with Extremely Small Number of Parameters","date":"2024-05-27","arxiv_id":"2405.17604","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lora-xs-low-rank-adaptation-with-extremely#ran","syntology_url":"https://syntology.ai/paper/2405.17604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17604"}},"official":{"repos":["mohammadrezabanaei/lora-xs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nuwats-mending-every-incomplete-time-series","slug":"nuwats-mending-every-incomplete-time-series","title":"NuwaTS: a Foundation Model Mending Every Incomplete Time Series","date":"2024-05-24","arxiv_id":"2405.15317","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuwats-mending-every-incomplete-time-series#ran","syntology_url":"https://syntology.ai/paper/2405.15317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15317"}},"official":{"repos":["chengyui/nuwats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gcondenser-benchmarking-graph-condensation","slug":"gcondenser-benchmarking-graph-condensation","title":"GCondenser: Benchmarking Graph Condensation","date":"2024-05-23","arxiv_id":"2405.14246","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gcondenser-benchmarking-graph-condensation#ran","syntology_url":"https://syntology.ai/paper/2405.14246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14246"}},"official":{"repos":["superallen13/GCondenser"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/androidworld-a-dynamic-benchmarking","slug":"androidworld-a-dynamic-benchmarking","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","date":"2024-05-23","arxiv_id":"2405.14573","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/androidworld-a-dynamic-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2405.14573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14573"}},"official":{"repos":["google-research/android_world"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-training-state-of-the","slug":"an-empirical-study-of-training-state-of-the","title":"An Empirical Study of Training State-of-the-Art LiDAR Segmentation Models","date":"2024-05-23","arxiv_id":"2405.14870","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-training-state-of-the#ran","syntology_url":"https://syntology.ai/paper/2405.14870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14870"}},"official":{"repos":["open-mmlab/mmdetection3d"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-fish-dataset-and-evaluation","slug":"benchmarking-fish-dataset-and-evaluation","title":"Benchmarking Fish Dataset and Evaluation Metric in Keypoint Detection -- Towards Precise Fish Morphological Assessment in Aquaculture Breeding","date":"2024-05-21","arxiv_id":"2405.12476","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/benchmarking-fish-dataset-and-evaluation#ran","syntology_url":"https://syntology.ai/paper/2405.12476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12476"}},"official":{"repos":["weizhenliubioinform/fish-phenotype-detect"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/polyglotoxicityprompts-multilingual","slug":"polyglotoxicityprompts-multilingual","title":"PolygloToxicityPrompts: Multilingual Evaluation of Neural Toxic Degeneration in Large Language Models","date":"2024-05-15","arxiv_id":"2405.09373","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polyglotoxicityprompts-multilingual#ran","syntology_url":"https://syntology.ai/paper/2405.09373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09373"}},"official":{"repos":["kpriyanshu256/polyglo-toxicity-prompts","rijgersberg/geitje"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-eeg-to-text-models-working","slug":"are-eeg-to-text-models-working","title":"Are EEG-to-Text Models Working?","date":"2024-05-10","arxiv_id":"2405.06459","repositories_listed":3,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/are-eeg-to-text-models-working#ran","syntology_url":"https://syntology.ai/paper/2405.06459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06459"}},"official":{"repos":["mikewangwzhl/eeg-to-text","neuspeech/eeg-to-text"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/openfactcheck-a-unified-framework-for","slug":"openfactcheck-a-unified-framework-for","title":"OpenFactCheck: Building, Benchmarking Customized Fact-Checking Systems and Evaluating the Factuality of Claims and LLMs","date":"2024-05-09","arxiv_id":"2405.05583","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openfactcheck-a-unified-framework-for#ran","syntology_url":"https://syntology.ai/paper/2405.05583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05583"}},"official":{"repos":["yuxiaw/openfactcheck"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-educational-program-repair","slug":"benchmarking-educational-program-repair","title":"Benchmarking Educational Program Repair","date":"2024-05-08","arxiv_id":"2405.05347","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-educational-program-repair#ran","syntology_url":"https://syntology.ai/paper/2405.05347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05347"}},"official":{"repos":["koutchemecharles/gaied_nips23"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/isearle-improving-textual-inversion-for-zero","slug":"isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","arxiv_id":"2405.02951","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/isearle-improving-textual-inversion-for-zero#ran","syntology_url":"https://syntology.ai/paper/2405.02951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02951"}},"official":{"repos":["miccunifi/circo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/position-paper-quo-vadis-unsupervised-time","slug":"position-paper-quo-vadis-unsupervised-time","title":"Position: Quo Vadis, Unsupervised Time Series Anomaly Detection?","date":"2024-05-04","arxiv_id":"2405.02678","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-paper-quo-vadis-unsupervised-time#ran","syntology_url":"https://syntology.ai/paper/2405.02678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02678"}},"official":{"repos":["ssarfraz/QuoVadisTAD"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-benchmark-leakage-in-large","slug":"benchmarking-benchmark-leakage-in-large","title":"Benchmarking Benchmark Leakage in Large Language Models","date":"2024-04-29","arxiv_id":"2404.18824","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-benchmark-leakage-in-large#ran","syntology_url":"https://syntology.ai/paper/2404.18824","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18824"}},"official":{"repos":["gair-nlp/benbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-critical-treatment-effect-bias-in","slug":"detecting-critical-treatment-effect-bias-in","title":"Detecting critical treatment effect bias in small subgroups","date":"2024-04-29","arxiv_id":"2404.18905","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/detecting-critical-treatment-effect-bias-in#ran","syntology_url":"https://syntology.ai/paper/2404.18905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18905"}},"official":{"repos":["jaabmar/kernel-test-bias"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/4dbinfer-a-4d-benchmarking-toolbox-for-graph","slug":"4dbinfer-a-4d-benchmarking-toolbox-for-graph","title":"4DBInfer: A 4D Benchmarking Toolbox for Graph-Centric Predictive Modeling on Relational DBs","date":"2024-04-28","arxiv_id":"2404.18209","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/4dbinfer-a-4d-benchmarking-toolbox-for-graph#ran","syntology_url":"https://syntology.ai/paper/2404.18209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18209"}},"official":{"repos":["awslabs/multi-table-benchmark"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seed-bench-2-plus-benchmarking-multimodal","slug":"seed-bench-2-plus-benchmarking-multimodal","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","date":"2024-04-25","arxiv_id":"2404.16790","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seed-bench-2-plus-benchmarking-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.16790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16790"}},"official":{"repos":["ailab-cvc/seed-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dpo-differential-reinforcement-learning-with","slug":"dpo-differential-reinforcement-learning-with","title":"DPO: A Differential and Pointwise Control Approach to Reinforcement Learning","date":"2024-04-24","arxiv_id":"2404.15617","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dpo-differential-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2404.15617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15617"}},"official":null}},{"url":"/paper/a-user-centric-benchmark-for-evaluating-large","slug":"a-user-centric-benchmark-for-evaluating-large","title":"A User-Centric Multi-Intent Benchmark for Evaluating Large Language Models","date":"2024-04-22","arxiv_id":"2404.13940","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-user-centric-benchmark-for-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2404.13940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13940"}},"official":{"repos":["alice1998/urs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tavgbench-benchmarking-text-to-audible-video","slug":"tavgbench-benchmarking-text-to-audible-video","title":"TAVGBench: Benchmarking Text to Audible-Video Generation","date":"2024-04-22","arxiv_id":"2404.14381","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":5,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tavgbench-benchmarking-text-to-audible-video#ran","syntology_url":"https://syntology.ai/paper/2404.14381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14381"}},"official":{"repos":["opennlplab/tavgbench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rexel-an-end-to-end-model-for-document-level","slug":"rexel-an-end-to-end-model-for-document-level","title":"REXEL: An End-to-end Model for Document-Level Relation Extraction and Entity Linking","date":"2024-04-19","arxiv_id":"2404.12788","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rexel-an-end-to-end-model-for-document-level#ran","syntology_url":"https://syntology.ai/paper/2404.12788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12788"}},"official":{"repos":["amazon-science/e2e-docie"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/stark-benchmarking-llm-retrieval-on-textual","slug":"stark-benchmarking-llm-retrieval-on-textual","title":"STaRK: Benchmarking LLM Retrieval on Textual and Relational Knowledge Bases","date":"2024-04-19","arxiv_id":"2404.13207","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stark-benchmarking-llm-retrieval-on-textual#ran","syntology_url":"https://syntology.ai/paper/2404.13207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13207"}},"official":{"repos":["snap-stanford/stark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/longembed-extending-embedding-models-for-long","slug":"longembed-extending-embedding-models-for-long","title":"LongEmbed: Extending Embedding Models for Long Context Retrieval","date":"2024-04-18","arxiv_id":"2404.12096","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/longembed-extending-embedding-models-for-long#ran","syntology_url":"https://syntology.ai/paper/2404.12096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12096"}},"official":{"repos":["dwzhu-pku/longembed"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmcode-evaluating-multi-modal-code-large","slug":"mmcode-evaluating-multi-modal-code-large","title":"MMCode: Benchmarking Multimodal Large Language Models for Code Generation with Visually Rich Programming Problems","date":"2024-04-15","arxiv_id":"2404.09486","repositories_listed":3,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mmcode-evaluating-multi-modal-code-large#ran","syntology_url":"https://syntology.ai/paper/2404.09486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09486"}},"official":{"repos":["happylkx/mmcode","likaixin2000/mmcode"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/nnu-net-revisited-a-call-for-rigorous","slug":"nnu-net-revisited-a-call-for-rigorous","title":"nnU-Net Revisited: A Call for Rigorous Validation in 3D Medical Image Segmentation","date":"2024-04-15","arxiv_id":"2404.09556","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nnu-net-revisited-a-call-for-rigorous#ran","syntology_url":"https://syntology.ai/paper/2404.09556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09556"}},"official":{"repos":["MIC-DKFZ/nnunet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"767f571ed9bf45072e586a4fccf91e7cf025284b382623a3fafb85582561aa51","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}