{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/13","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":56,"rows_per_page":100,"rows":[1201,1300],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/12","next":"/task/benchmarking/papers/14","papers":[{"url":"/paper/simulation-based-benchmarking-for-causal","slug":"simulation-based-benchmarking-for-causal","title":"Simulation-based Benchmarking for Causal Structure Learning in Gene Perturbation Experiments","date":"2024-07-08","arxiv_id":"2407.06015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simulation-based-benchmarking-for-causal#ran","syntology_url":"https://syntology.ai/paper/2407.06015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06015"}},"official":{"repos":["luka-kovacevic/causalregnet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-00001","slug":"2408-00001","title":"Replication in Visual Diffusion Models: A Survey and Outlook","date":"2024-07-07","arxiv_id":"2408.00001","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-the-effectiveness-of-graph","slug":"rethinking-the-effectiveness-of-graph","title":"Rethinking the Effectiveness of Graph Classification Datasets in Benchmarks for Assessing GNNs","date":"2024-07-06","arxiv_id":"2407.04999","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rethinking-the-effectiveness-of-graph#ran","syntology_url":"https://syntology.ai/paper/2407.04999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04999"}},"official":{"repos":["ICLab4DL/GNNBenchEffectiveness"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-structure-based-three","slug":"benchmarking-structure-based-three","title":"Benchmarking structure-based three-dimensional molecular generative models using GenBench3D: ligand conformation quality matters","date":"2024-07-05","arxiv_id":"2407.04424","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-structure-based-three#ran","syntology_url":"https://syntology.ai/paper/2407.04424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04424"}},"official":{"repos":["bbaillif/genbench3d"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmark-on-drug-target-interaction-modeling","slug":"benchmark-on-drug-target-interaction-modeling","title":"Benchmark on Drug Target Interaction Modeling from a Structure Perspective","date":"2024-07-04","arxiv_id":"2407.04055","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-complex-instruction-following","slug":"benchmarking-complex-instruction-following","title":"Benchmarking Complex Instruction-Following with Multiple Constraints Composition","date":"2024-07-04","arxiv_id":"2407.03978","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-complex-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2407.03978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03978"}},"official":{"repos":["thu-coai/complexbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/craftium-an-extensible-framework-for-creating","slug":"craftium-an-extensible-framework-for-creating","title":"Craftium: An Extensible Framework for Creating Reinforcement Learning Environments","date":"2024-07-04","arxiv_id":"2407.03969","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/craftium-an-extensible-framework-for-creating#ran","syntology_url":"https://syntology.ai/paper/2407.03969","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03969"}},"official":{"repos":["mikelma/craftium"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coir-a-comprehensive-benchmark-for-code","slug":"coir-a-comprehensive-benchmark-for-code","title":"CoIR: A Comprehensive Benchmark for Code Information Retrieval Models","date":"2024-07-03","arxiv_id":"2407.02883","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/coir-a-comprehensive-benchmark-for-code#ran","syntology_url":"https://syntology.ai/paper/2407.02883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02883"}},"official":{"repos":["coir-team/coir"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/emotion-and-intent-joint-understanding-in","slug":"emotion-and-intent-joint-understanding-in","title":"Emotion and Intent Joint Understanding in Multimodal Conversation: A Benchmarking Dataset","date":"2024-07-03","arxiv_id":"2407.02751","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emotion-and-intent-joint-understanding-in#ran","syntology_url":"https://syntology.ai/paper/2407.02751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02751"}},"official":{"repos":["mc-eiu/mc-eiu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gracore-benchmarking-graph-comprehension-and","slug":"gracore-benchmarking-graph-comprehension-and","title":"GraCoRe: Benchmarking Graph Comprehension and Complex Reasoning in Large Language Models","date":"2024-07-03","arxiv_id":"2407.02936","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gracore-benchmarking-graph-comprehension-and#ran","syntology_url":"https://syntology.ai/paper/2407.02936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02936"}},"official":{"repos":["zikeyuan/gracore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/social-bias-in-large-language-models-for","slug":"social-bias-in-large-language-models-for","title":"Social Bias in Large Language Models For Bangla: An Empirical Study on Gender and Religious Bias","date":"2024-07-03","arxiv_id":"2407.03536","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-ability-of-llms-to-solve","slug":"evaluating-the-ability-of-llms-to-solve","title":"Evaluating the Ability of LLMs to Solve Semantics-Aware Process Mining Tasks","date":"2024-07-02","arxiv_id":"2407.02310","repositories_listed":1,"syntology":null},{"url":"/paper/occlusion-aware-seamless-segmentation","slug":"occlusion-aware-seamless-segmentation","title":"Occlusion-Aware Seamless Segmentation","date":"2024-07-02","arxiv_id":"2407.02182","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/occlusion-aware-seamless-segmentation#ran","syntology_url":"https://syntology.ai/paper/2407.02182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02182"}},"official":{"repos":["yihong-97/oass"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-agents-that-matter","slug":"ai-agents-that-matter","title":"AI Agents That Matter","date":"2024-07-01","arxiv_id":"2407.01502","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai-agents-that-matter#ran","syntology_url":"https://syntology.ai/paper/2407.01502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01502"}},"official":{"repos":["benediktstroebl/agent-evals"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-predictive-coding-networks-made","slug":"benchmarking-predictive-coding-networks-made","title":"Benchmarking Predictive Coding Networks -- Made Simple","date":"2024-07-01","arxiv_id":"2407.01163","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-predictive-coding-networks-made#ran","syntology_url":"https://syntology.ai/paper/2407.01163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01163"}},"official":{"repos":["liukidar/pcax"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bergen-a-benchmarking-library-for-retrieval","slug":"bergen-a-benchmarking-library-for-retrieval","title":"BERGEN: A Benchmarking Library for Retrieval-Augmented Generation","date":"2024-07-01","arxiv_id":"2407.01102","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bergen-a-benchmarking-library-for-retrieval#ran","syntology_url":"https://syntology.ai/paper/2407.01102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01102"}},"official":{"repos":["naver/bergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fairmedfm-fairness-benchmarking-for-medical","slug":"fairmedfm-fairness-benchmarking-for-medical","title":"FairMedFM: Fairness Benchmarking for Medical Imaging Foundation Models","date":"2024-07-01","arxiv_id":"2407.00983","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fairmedfm-fairness-benchmarking-for-medical#ran","syntology_url":"https://syntology.ai/paper/2407.00983","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00983"}},"official":{"repos":["FairMedFM/FairMedFM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmlongbench-doc-benchmarking-long-context","slug":"mmlongbench-doc-benchmarking-long-context","title":"MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations","date":"2024-07-01","arxiv_id":"2407.01523","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmlongbench-doc-benchmarking-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.01523","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01523"}},"official":null}},{"url":"/paper/mobile-bench-an-evaluation-benchmark-for-llm","slug":"mobile-bench-an-evaluation-benchmark-for-llm","title":"Mobile-Bench: An Evaluation Benchmark for LLM-based Mobile Agents","date":"2024-07-01","arxiv_id":"2407.00993","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mobile-bench-an-evaluation-benchmark-for-llm#ran","syntology_url":"https://syntology.ai/paper/2407.00993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00993"}},"official":{"repos":["XiaoMi/MobileBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-common-flaws-in-the-evaluation-of","slug":"overcoming-common-flaws-in-the-evaluation-of","title":"Overcoming Common Flaws in the Evaluation of Selective Classification Systems","date":"2024-07-01","arxiv_id":"2407.01032","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/overcoming-common-flaws-in-the-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2407.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01032"}},"official":{"repos":["iml-dkfz/fd-shifts"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/reinvestigating-the-r2-indicator-achieving","slug":"reinvestigating-the-r2-indicator-achieving","title":"Reinvestigating the R2 Indicator: Achieving Pareto Compliance by Integration","date":"2024-07-01","arxiv_id":"2407.01504","repositories_listed":1,"syntology":null},{"url":"/paper/grapharena-benchmarking-large-language-models","slug":"grapharena-benchmarking-large-language-models","title":"GraphArena: Benchmarking Large Language Models on Graph Computational Problems","date":"2024-06-29","arxiv_id":"2407.00379","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grapharena-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.00379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00379"}},"official":{"repos":["squareroot3/grapharena"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/iampcn-a-deep-learning-approach-for","slug":"iampcn-a-deep-learning-approach-for","title":"iAMPCN: a deep-learning approach for identifying antimicrobial peptides and their functional activities","date":"2024-06-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unigen-a-unified-framework-for-textual","slug":"unigen-a-unified-framework-for-textual","title":"UniGen: A Unified Framework for Textual Dataset Generation Using Large Language Models","date":"2024-06-27","arxiv_id":"2406.18966","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unigen-a-unified-framework-for-textual#ran","syntology_url":"https://syntology.ai/paper/2406.18966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18966"}},"official":{"repos":["howiehwong/unigen"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-foundation-world-models-for","slug":"multimodal-foundation-world-models-for","title":"GenRL: Multimodal-foundation world models for generalization in embodied agents","date":"2024-06-26","arxiv_id":"2406.18043","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multimodal-foundation-world-models-for#ran","syntology_url":"https://syntology.ai/paper/2406.18043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18043"}},"official":{"repos":["mazpie/genrl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-thorough-performance-benchmarking-on","slug":"a-thorough-performance-benchmarking-on","title":"A Thorough Performance Benchmarking on Lightweight Embedding-based Recommender Systems","date":"2024-06-25","arxiv_id":"2406.17335","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-deep-learning-models-on-nvidia","slug":"benchmarking-deep-learning-models-on-nvidia","title":"Benchmarking Deep Learning Models on NVIDIA Jetson Nano for Real-Time Systems: An Empirical Investigation","date":"2024-06-25","arxiv_id":"2406.17749","repositories_listed":1,"syntology":null},{"url":"/paper/depth-driven-geometric-prompt-learning-for","slug":"depth-driven-geometric-prompt-learning-for","title":"Depth-Driven Geometric Prompt Learning for Laparoscopic Liver Landmark Detection","date":"2024-06-25","arxiv_id":"2406.17858","repositories_listed":1,"syntology":null},{"url":"/paper/inherent-challenges-of-post-hoc-membership","slug":"inherent-challenges-of-post-hoc-membership","title":"SoK: Membership Inference Attacks on LLMs are Rushing Nowhere (and How to Fix It)","date":"2024-06-25","arxiv_id":"2406.17975","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inherent-challenges-of-post-hoc-membership#ran","syntology_url":"https://syntology.ai/paper/2406.17975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17975"}},"official":{"repos":["computationalprivacy/mia_llms_benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leave-no-document-behind-benchmarking-long","slug":"leave-no-document-behind-benchmarking-long","title":"Leave No Document Behind: Benchmarking Long-Context LLMs with Extended Multi-Doc QA","date":"2024-06-25","arxiv_id":"2406.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/leave-no-document-behind-benchmarking-long#ran","syntology_url":"https://syntology.ai/paper/2406.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17419"}},"official":{"repos":["mozerwang/loong"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mattext-do-language-models-need-more-than","slug":"mattext-do-language-models-need-more-than","title":"MatText: Do Language Models Need More than Text & Scale for Materials Modeling?","date":"2024-06-25","arxiv_id":"2406.17295","repositories_listed":1,"syntology":null},{"url":"/paper/towards-efficient-and-scalable-training-of","slug":"towards-efficient-and-scalable-training-of","title":"Towards Efficient and Scalable Training of Differentially Private Deep Learning","date":"2024-06-25","arxiv_id":"2406.17298","repositories_listed":1,"syntology":null},{"url":"/paper/varbench-robust-language-model-benchmarking","slug":"varbench-robust-language-model-benchmarking","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","date":"2024-06-25","arxiv_id":"2406.17681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/varbench-robust-language-model-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.17681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17681"}},"official":{"repos":["qbetterk/VarBench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autodetect-towards-a-unified-framework-for","slug":"autodetect-towards-a-unified-framework-for","title":"AutoDetect: Towards a Unified Framework for Automated Weakness Detection in Large Language Models","date":"2024-06-24","arxiv_id":"2406.16714","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-mortality-risk-prediction-from","slug":"benchmarking-mortality-risk-prediction-from","title":"A Closer Look at Mortality Risk Prediction from Electrocardiograms","date":"2024-06-24","arxiv_id":"2406.17002","repositories_listed":1,"syntology":null},{"url":"/paper/dreambench-a-human-aligned-benchmark-for","slug":"dreambench-a-human-aligned-benchmark-for","title":"DreamBench++: A Human-Aligned Benchmark for Personalized Image Generation","date":"2024-06-24","arxiv_id":"2406.16855","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dreambench-a-human-aligned-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.16855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16855"}},"official":{"repos":["yuangpeng/dreambench_plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-diffusion-models-for-enhancing","slug":"fine-tuning-diffusion-models-for-enhancing","title":"FaceScore: Benchmarking and Enhancing Face Quality in Human Generation","date":"2024-06-24","arxiv_id":"2406.17100","repositories_listed":1,"syntology":null},{"url":"/paper/from-perfect-to-noisy-world-simulation","slug":"from-perfect-to-noisy-world-simulation","title":"From Perfect to Noisy World Simulation: Customizable Embodied Multi-modal Perturbations for SLAM Robustness Benchmarking","date":"2024-06-24","arxiv_id":"2406.16850","repositories_listed":1,"syntology":null},{"url":"/paper/general-binding-affinity-guidance-for","slug":"general-binding-affinity-guidance-for","title":"General Binding Affinity Guidance for Diffusion Models in Structure-Based Drug Design","date":"2024-06-24","arxiv_id":"2406.16821","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/general-binding-affinity-guidance-for#ran","syntology_url":"https://syntology.ai/paper/2406.16821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16821"}},"official":{"repos":["ask-berkeley/badger-sbdd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hest-1k-a-dataset-for-spatial-transcriptomics","slug":"hest-1k-a-dataset-for-spatial-transcriptomics","title":"HEST-1k: A Dataset for Spatial Transcriptomics and Histology Image Analysis","date":"2024-06-23","arxiv_id":"2406.16192","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hest-1k-a-dataset-for-spatial-transcriptomics#ran","syntology_url":"https://syntology.ai/paper/2406.16192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16192"}},"official":{"repos":["mahmoodlab/hest"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-open-respiratory-acoustic-foundation","slug":"towards-open-respiratory-acoustic-foundation","title":"Towards Open Respiratory Acoustic Foundation Models: Pretraining and Benchmarking","date":"2024-06-23","arxiv_id":"2406.16148","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-open-respiratory-acoustic-foundation#ran","syntology_url":"https://syntology.ai/paper/2406.16148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16148"}},"official":{"repos":["evelyn0414/opera"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/metagreen-meta-learning-inspired-transformer","slug":"metagreen-meta-learning-inspired-transformer","title":"MetaGreen: Meta-Learning Inspired Transformer Selection for Green Semantic Communication","date":"2024-06-22","arxiv_id":"2406.16962","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-retinal-blood-vessel","slug":"benchmarking-retinal-blood-vessel","title":"Benchmarking Retinal Blood Vessel Segmentation Models for Cross-Dataset and Cross-Disease Generalization","date":"2024-06-21","arxiv_id":"2406.14994","repositories_listed":1,"syntology":null},{"url":"/paper/deciphering-the-definition-of-adversarial","slug":"deciphering-the-definition-of-adversarial","title":"Deciphering the Definition of Adversarial Robustness for post-hoc OOD Detectors","date":"2024-06-21","arxiv_id":"2406.15104","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/deciphering-the-definition-of-adversarial#ran","syntology_url":"https://syntology.ai/paper/2406.15104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15104"}},"official":{"repos":["adverml/advopenood"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genotex-a-benchmark-for-evaluating-llm-based","slug":"genotex-a-benchmark-for-evaluating-llm-based","title":"GenoTEX: An LLM Agent Benchmark for Automated Gene Expression Data Analysis","date":"2024-06-21","arxiv_id":"2406.15341","repositories_listed":1,"syntology":null},{"url":"/paper/six-cd-benchmarking-concept-removals-for","slug":"six-cd-benchmarking-concept-removals-for","title":"Six-CD: Benchmarking Concept Removals for Benign Text-to-image Diffusion Models","date":"2024-06-21","arxiv_id":"2406.14855","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmarking-study-of-kolmogorov-arnold","slug":"a-benchmarking-study-of-kolmogorov-arnold","title":"A Benchmarking Study of Kolmogorov-Arnold Networks on Tabular Data","date":"2024-06-20","arxiv_id":"2406.14529","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-benchmarking-study-of-kolmogorov-arnold#ran","syntology_url":"https://syntology.ai/paper/2406.14529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14529"}},"official":{"repos":["eleonorapoeta/benchmarking-kan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/african-or-european-swallow-benchmarking","slug":"african-or-european-swallow-benchmarking","title":"African or European Swallow? Benchmarking Large Vision-Language Models for Fine-Grained Object Classification","date":"2024-06-20","arxiv_id":"2406.14496","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/african-or-european-swallow-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.14496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14496"}},"official":{"repos":["gregor-ge/foci-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-optimism-exploration-with-partially","slug":"beyond-optimism-exploration-with-partially","title":"Beyond Optimism: Exploration With Partially Observable Rewards","date":"2024-06-20","arxiv_id":"2406.13909","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":10,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/beyond-optimism-exploration-with-partially#ran","syntology_url":"https://syntology.ai/paper/2406.13909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13909"}},"official":{"repos":["AmiiThinks/mon_mdp_neurips24"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/fairx-a-comprehensive-benchmarking-tool-for","slug":"fairx-a-comprehensive-benchmarking-tool-for","title":"FairX: A comprehensive benchmarking tool for model analysis using fairness, utility, and explainability","date":"2024-06-20","arxiv_id":"2406.14281","repositories_listed":1,"syntology":null},{"url":"/paper/the-elusive-pursuit-of-replicating-pate-gan","slug":"the-elusive-pursuit-of-replicating-pate-gan","title":"The Elusive Pursuit of Reproducing PATE-GAN: Benchmarking, Auditing, Debugging","date":"2024-06-20","arxiv_id":"2406.13985","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-elusive-pursuit-of-replicating-pate-gan#ran","syntology_url":"https://syntology.ai/paper/2406.13985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13985"}},"official":{"repos":["spalabucr/pategan-audit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-truthful-multilingual-large-language","slug":"towards-truthful-multilingual-large-language","title":"Selected Languages are All You Need for Cross-lingual Truthfulness Transfer","date":"2024-06-20","arxiv_id":"2406.14434","repositories_listed":1,"syntology":null},{"url":"/paper/weather-5k-a-large-scale-global-station","slug":"weather-5k-a-large-scale-global-station","title":"How far are today's time-series models from real-world weather forecasting applications?","date":"2024-06-20","arxiv_id":"2406.14399","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":15,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 15 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/weather-5k-a-large-scale-global-station#ran","syntology_url":"https://syntology.ai/paper/2406.14399","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14399"}},"official":{"repos":["taohan10200/weather-5k"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":15,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/behonest-benchmarking-honesty-of-large","slug":"behonest-benchmarking-honesty-of-large","title":"BeHonest: Benchmarking Honesty in Large Language Models","date":"2024-06-19","arxiv_id":"2406.13261","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-unsupervised-online-ids-for","slug":"benchmarking-unsupervised-online-ids-for","title":"Benchmarking Unsupervised Online IDS for Masquerade Attacks in CAN","date":"2024-06-19","arxiv_id":"2406.13778","repositories_listed":1,"syntology":null},{"url":"/paper/m4fog-a-global-multi-regional-multi-modal-and","slug":"m4fog-a-global-multi-regional-multi-modal-and","title":"M4Fog: A Global Multi-Regional, Multi-Modal, and Multi-Stage Dataset for Marine Fog Detection and Forecasting to Bridge Ocean and Atmosphere","date":"2024-06-19","arxiv_id":"2406.13317","repositories_listed":1,"syntology":null},{"url":"/paper/mama-mia-a-large-scale-multi-center-breast","slug":"mama-mia-a-large-scale-multi-center-breast","title":"A large-scale multicenter breast cancer DCE-MRI benchmark dataset with expert segmentations","date":"2024-06-19","arxiv_id":"2406.13844","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mama-mia-a-large-scale-multi-center-breast#ran","syntology_url":"https://syntology.ai/paper/2406.13844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13844"}},"official":{"repos":["lidiagarrucho/mama-mia"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-benchmarking-of-large-multimodal","slug":"automatic-benchmarking-of-large-multimodal","title":"Automatic benchmarking of large multimodal models via iterative experiment programming","date":"2024-06-18","arxiv_id":"2406.12321","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multi-image-understanding-in","slug":"benchmarking-multi-image-understanding-in","title":"Benchmarking Multi-Image Understanding in Vision and Language Models: Perception, Knowledge, Reasoning, and Multi-Hop Reasoning","date":"2024-06-18","arxiv_id":"2406.12742","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-multi-image-understanding-in#ran","syntology_url":"https://syntology.ai/paper/2406.12742","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12742"}},"official":{"repos":["dtennant/mirb_eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/geobench-benchmarking-and-analyzing-monocular","slug":"geobench-benchmarking-and-analyzing-monocular","title":"GeoBench: Benchmarking and Analyzing Monocular Geometry Estimation Models","date":"2024-06-18","arxiv_id":"2406.12671","repositories_listed":1,"syntology":null},{"url":"/paper/olympicarena-benchmarking-multi-discipline","slug":"olympicarena-benchmarking-multi-discipline","title":"OlympicArena: Benchmarking Multi-discipline Cognitive Reasoning for Superintelligent AI","date":"2024-06-18","arxiv_id":"2406.12753","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/olympicarena-benchmarking-multi-discipline#ran","syntology_url":"https://syntology.ai/paper/2406.12753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12753"}},"official":{"repos":["gair-nlp/olympicarena"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ubench-benchmarking-uncertainty-in-large","slug":"ubench-benchmarking-uncertainty-in-large","title":"UBENCH: Benchmarking Uncertainty in Large Language Models with Multiple Choice Questions","date":"2024-06-18","arxiv_id":"2406.12784","repositories_listed":1,"syntology":null},{"url":"/paper/webcanvas-benchmarking-web-agents-in-online","slug":"webcanvas-benchmarking-web-agents-in-online","title":"WebCanvas: Benchmarking Web Agents in Online Environments","date":"2024-06-18","arxiv_id":"2406.12373","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/webcanvas-benchmarking-web-agents-in-online#ran","syntology_url":"https://syntology.ai/paper/2406.12373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12373"}},"official":{"repos":["imeanai/webcanvas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-true-healthcare","slug":"are-large-language-models-true-healthcare","title":"Are Large Language Models True Healthcare Jacks-of-All-Trades? Benchmarking Across Health Professions Beyond Physician Exams","date":"2024-06-17","arxiv_id":"2406.11328","repositories_listed":1,"syntology":null},{"url":"/paper/gecobench-a-gender-controlled-text-dataset","slug":"gecobench-a-gender-controlled-text-dataset","title":"GECOBench: A Gender-Controlled Text Dataset and Benchmark for Quantifying Biases in Explanations","date":"2024-06-17","arxiv_id":"2406.11547","repositories_listed":1,"syntology":null},{"url":"/paper/job-sdf-a-multi-granularity-dataset-for-job","slug":"job-sdf-a-multi-granularity-dataset-for-job","title":"Job-SDF: A Multi-Granularity Dataset for Job Skill Demand Forecasting and Benchmarking","date":"2024-06-17","arxiv_id":"2406.11920","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/job-sdf-a-multi-granularity-dataset-for-job#ran","syntology_url":"https://syntology.ai/paper/2406.11920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11920"}},"official":{"repos":["job-sdf/benchmark"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mfc-bench-benchmarking-multimodal-fact","slug":"mfc-bench-benchmarking-multimodal-fact","title":"MFC-Bench: Benchmarking Multimodal Fact-Checking with Large Vision-Language Models","date":"2024-06-17","arxiv_id":"2406.11288","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-needle-in-a-haystack-benchmarking","slug":"multimodal-needle-in-a-haystack-benchmarking","title":"Multimodal Needle in a Haystack: Benchmarking Long-Context Capability of Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11230","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-needle-in-a-haystack-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.11230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11230"}},"official":{"repos":["wang-ml-lab/multimodal-needle-in-a-haystack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/repliqa-a-question-answering-dataset-for","slug":"repliqa-a-question-answering-dataset-for","title":"RepLiQA: A Question-Answering Dataset for Benchmarking LLMs on Unseen Reference Content","date":"2024-06-17","arxiv_id":"2406.11811","repositories_listed":1,"syntology":null},{"url":"/paper/standardizing-structural-causal-models","slug":"standardizing-structural-causal-models","title":"Standardizing Structural Causal Models","date":"2024-06-17","arxiv_id":"2406.11601","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/standardizing-structural-causal-models#ran","syntology_url":"https://syntology.ai/paper/2406.11601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11601"}},"official":{"repos":["werkaaa/iscm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-label-noise-in-instance","slug":"benchmarking-label-noise-in-instance","title":"Benchmarking Label Noise in Instance Segmentation: Spatial Noise Matters","date":"2024-06-16","arxiv_id":"2406.10891","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-label-noise-in-instance#ran","syntology_url":"https://syntology.ai/paper/2406.10891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10891"}},"official":{"repos":["eden500/Noisy-Labels-Instance-Segmentation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rupbench-benchmarking-reasoning-under","slug":"rupbench-benchmarking-reasoning-under","title":"RUPBench: Benchmarking Reasoning Under Perturbations for Robustness Evaluation in Large Language Models","date":"2024-06-16","arxiv_id":"2406.11020","repositories_listed":1,"syntology":null},{"url":"/paper/rwku-benchmarking-real-world-knowledge","slug":"rwku-benchmarking-real-world-knowledge","title":"RWKU: Benchmarking Real-World Knowledge Unlearning for Large Language Models","date":"2024-06-16","arxiv_id":"2406.10890","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rwku-benchmarking-real-world-knowledge#ran","syntology_url":"https://syntology.ai/paper/2406.10890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10890"}},"official":{"repos":["jinzhuoran/rwku"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-gpu-accelerated-large-scale-simulator-for","slug":"a-gpu-accelerated-large-scale-simulator-for","title":"A GPU-accelerated Large-scale Simulator for Transportation System Optimization Benchmarking","date":"2024-06-15","arxiv_id":"2406.10661","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-children-s-asr-with-supervised","slug":"benchmarking-children-s-asr-with-supervised","title":"Benchmarking Children's ASR with Supervised and Self-supervised Speech Foundation Models","date":"2024-06-15","arxiv_id":"2406.10507","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-children-s-asr-with-supervised#ran","syntology_url":"https://syntology.ai/paper/2406.10507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10507"}},"official":{"repos":["Diamondfan/SPAPL_KidsASR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-spectral-graph-neural-networks-a","slug":"benchmarking-spectral-graph-neural-networks-a","title":"Benchmarking Spectral Graph Neural Networks: A Comprehensive Study on Effectiveness and Efficiency","date":"2024-06-14","arxiv_id":"2406.09675","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-spectral-graph-neural-networks-a#ran","syntology_url":"https://syntology.ai/paper/2406.09675","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09675"}},"official":{"repos":["gdmnl/spectral-gnn-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-slow-signs-in-high-fidelity-model","slug":"beyond-slow-signs-in-high-fidelity-model","title":"Beyond Slow Signs in High-fidelity Model Extraction","date":"2024-06-14","arxiv_id":"2406.10011","repositories_listed":1,"syntology":null},{"url":"/paper/climretrieve-a-benchmarking-dataset-for","slug":"climretrieve-a-benchmarking-dataset-for","title":"ClimRetrieve: A Benchmarking Dataset for Information Retrieval from Corporate Climate Disclosures","date":"2024-06-14","arxiv_id":"2406.09818","repositories_listed":1,"syntology":null},{"url":"/paper/luma-a-benchmark-dataset-for-learning-from","slug":"luma-a-benchmark-dataset-for-learning-from","title":"LUMA: A Benchmark Dataset for Learning from Uncertain and Multimodal Data","date":"2024-06-14","arxiv_id":"2406.09864","repositories_listed":1,"syntology":null},{"url":"/paper/sciex-benchmarking-large-language-models-on","slug":"sciex-benchmarking-large-language-models-on","title":"SciEx: Benchmarking Large Language Models on Scientific Exams with Human Expert Grading and Automatic Grading","date":"2024-06-14","arxiv_id":"2406.10421","repositories_listed":1,"syntology":null},{"url":"/paper/bts-building-timeseries-dataset-empowering","slug":"bts-building-timeseries-dataset-empowering","title":"BTS: Building Timeseries Dataset: Empowering Large-Scale Building Analytics","date":"2024-06-13","arxiv_id":"2406.08990","repositories_listed":1,"syntology":null},{"url":"/paper/defan-definitive-answer-dataset-for-llms","slug":"defan-definitive-answer-dataset-for-llms","title":"DefAn: Definitive Answer Dataset for LLMs Hallucination Evaluation","date":"2024-06-13","arxiv_id":"2406.09155","repositories_listed":1,"syntology":null},{"url":"/paper/ecbd-evidence-centered-benchmark-design-for","slug":"ecbd-evidence-centered-benchmark-design-for","title":"ECBD: Evidence-Centered Benchmark Design for NLP","date":"2024-06-13","arxiv_id":"2406.08723","repositories_listed":1,"syntology":null},{"url":"/paper/needle-in-a-video-haystack-a-scalable","slug":"needle-in-a-video-haystack-a-scalable","title":"Needle In A Video Haystack: A Scalable Synthetic Evaluator for Video MLLMs","date":"2024-06-13","arxiv_id":"2406.09367","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/needle-in-a-video-haystack-a-scalable#ran","syntology_url":"https://syntology.ai/paper/2406.09367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09367"}},"official":{"repos":["joez17/videoniah"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sciknoweval-evaluating-multi-level-scientific","slug":"sciknoweval-evaluating-multi-level-scientific","title":"SciKnowEval: Evaluating Multi-level Scientific Knowledge of Large Language Models","date":"2024-06-13","arxiv_id":"2406.09098","repositories_listed":1,"syntology":null},{"url":"/paper/sr-caco-2-a-dataset-for-confocal-fluorescence","slug":"sr-caco-2-a-dataset-for-confocal-fluorescence","title":"SR-CACO-2: A Dataset for Confocal Fluorescence Microscopy Image Super-Resolution","date":"2024-06-13","arxiv_id":"2406.09168","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sr-caco-2-a-dataset-for-confocal-fluorescence#ran","syntology_url":"https://syntology.ai/paper/2406.09168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09168"}},"official":{"repos":["sbelharbi/sr-caco-2"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/streambench-towards-benchmarking-continuous","slug":"streambench-towards-benchmarking-continuous","title":"StreamBench: Towards Benchmarking Continuous Improvement of Language Agents","date":"2024-06-13","arxiv_id":"2406.08747","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/streambench-towards-benchmarking-continuous#ran","syntology_url":"https://syntology.ai/paper/2406.08747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08747"}},"official":{"repos":["stream-bench/stream-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/causality-for-tabular-data-synthesis-a-high","slug":"causality-for-tabular-data-synthesis-a-high","title":"Causality for Tabular Data Synthesis: A High-Order Structure Causal Benchmark Framework","date":"2024-06-12","arxiv_id":"2406.08311","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causality-for-tabular-data-synthesis-a-high#ran","syntology_url":"https://syntology.ai/paper/2406.08311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08311"}},"official":{"repos":["turuibo/cautabbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/examining-post-training-quantization-for","slug":"examining-post-training-quantization-for","title":"Examining Post-Training Quantization for Mixture-of-Experts: A Benchmark","date":"2024-06-12","arxiv_id":"2406.08155","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/examining-post-training-quantization-for#ran","syntology_url":"https://syntology.ai/paper/2406.08155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08155"}},"official":{"repos":["unites-lab/moe-quantization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/language-model-council-benchmarking","slug":"language-model-council-benchmarking","title":"Language Model Council: Democratically Benchmarking Foundation Models on Highly Subjective Tasks","date":"2024-06-12","arxiv_id":"2406.08598","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-to-disentangle","slug":"reinforcement-learning-to-disentangle","title":"Reinforcement Learning to Disentangle Multiqubit Quantum States from Partial Observations","date":"2024-06-12","arxiv_id":"2406.07884","repositories_listed":1,"syntology":null},{"url":"/paper/tc-bench-benchmarking-temporal","slug":"tc-bench-benchmarking-temporal","title":"TC-Bench: Benchmarking Temporal Compositionality in Text-to-Video and Image-to-Video Generation","date":"2024-06-12","arxiv_id":"2406.08656","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tc-bench-benchmarking-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.08656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08656"}},"official":{"repos":["weixi-feng/tc-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/audiomarkbench-benchmarking-robustness-of","slug":"audiomarkbench-benchmarking-robustness-of","title":"AudioMarkBench: Benchmarking Robustness of Audio Watermarking","date":"2024-06-11","arxiv_id":"2406.06979","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiomarkbench-benchmarking-robustness-of#ran","syntology_url":"https://syntology.ai/paper/2406.06979","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06979"}},"official":{"repos":["moyangkuo/audiomarkbench"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-vision-language-contrastive","slug":"benchmarking-vision-language-contrastive","title":"Benchmarking Vision-Language Contrastive Methods for Medical Representation Learning","date":"2024-06-11","arxiv_id":"2406.07450","repositories_listed":1,"syntology":null},{"url":"/paper/rad-a-comprehensive-dataset-for-benchmarking","slug":"rad-a-comprehensive-dataset-for-benchmarking","title":"RAD: A Comprehensive Dataset for Benchmarking the Robustness of Image Anomaly Detection","date":"2024-06-11","arxiv_id":"2406.07176","repositories_listed":1,"syntology":null},{"url":"/paper/can-ai-beat-undergraduates-in-entry-level","slug":"can-ai-beat-undergraduates-in-entry-level","title":"JavaBench: A Benchmark of Object-Oriented Code Generation for Evaluating Large Language Models","date":"2024-06-10","arxiv_id":"2406.12902","repositories_listed":1,"syntology":null},{"url":"/paper/discoveryworld-a-virtual-environment-for","slug":"discoveryworld-a-virtual-environment-for","title":"DISCOVERYWORLD: A Virtual Environment for Developing and Evaluating Automated Scientific Discovery Agents","date":"2024-06-10","arxiv_id":"2406.06769","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discoveryworld-a-virtual-environment-for#ran","syntology_url":"https://syntology.ai/paper/2406.06769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06769"}},"official":{"repos":["allenai/discoveryworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-generalization-of-neural-vehicle","slug":"improving-generalization-of-neural-vehicle","title":"Improving Generalization of Neural Vehicle Routing Problem Solvers Through the Lens of Model Architecture","date":"2024-06-10","arxiv_id":"2406.06652","repositories_listed":1,"syntology":null},{"url":"/paper/interspeech-2009-emotion-challenge-revisited","slug":"interspeech-2009-emotion-challenge-revisited","title":"INTERSPEECH 2009 Emotion Challenge Revisited: Benchmarking 15 Years of Progress in Speech Emotion Recognition","date":"2024-06-10","arxiv_id":"2406.06401","repositories_listed":1,"syntology":null},{"url":"/paper/embspatial-bench-benchmarking-spatial","slug":"embspatial-bench-benchmarking-spatial","title":"EmbSpatial-Bench: Benchmarking Spatial Understanding for Embodied Tasks with Large Vision-Language Models","date":"2024-06-09","arxiv_id":"2406.05756","repositories_listed":1,"syntology":null}],"record_sha256":"4a5de0a78b7f832f42d63cae4b71e288061049b265340cf442e1d9a439741070","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}