{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/11","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":56,"rows_per_page":100,"rows":[1001,1100],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/10","next":"/task/benchmarking/papers/12","papers":[{"url":"/paper/comprehensive-benchmarking-of-large-language","slug":"comprehensive-benchmarking-of-large-language","title":"Comprehensive benchmarking of large language models for RNA secondary structure prediction","date":"2024-10-21","arxiv_id":"2410.16212","repositories_listed":1,"syntology":null},{"url":"/paper/multi-if-benchmarking-llms-on-multi-turn-and","slug":"multi-if-benchmarking-llms-on-multi-turn-and","title":"Multi-IF: Benchmarking LLMs on Multi-Turn and Multilingual Instructions Following","date":"2024-10-21","arxiv_id":"2410.15553","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-if-benchmarking-llms-on-multi-turn-and#ran","syntology_url":"https://syntology.ai/paper/2410.15553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15553"}},"official":{"repos":["facebookresearch/Multi-IF"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rm-bench-benchmarking-reward-models-of","slug":"rm-bench-benchmarking-reward-models-of","title":"RM-Bench: Benchmarking Reward Models of Language Models with Subtlety and Style","date":"2024-10-21","arxiv_id":"2410.16184","repositories_listed":1,"syntology":null},{"url":"/paper/flexmol-a-flexible-toolkit-for-benchmarking","slug":"flexmol-a-flexible-toolkit-for-benchmarking","title":"FlexMol: A Flexible Toolkit for Benchmarking Molecular Relational Learning","date":"2024-10-19","arxiv_id":"2410.15010","repositories_listed":1,"syntology":null},{"url":"/paper/intersectionzoo-eco-driving-for-benchmarking","slug":"intersectionzoo-eco-driving-for-benchmarking","title":"IntersectionZoo: Eco-driving for Benchmarking Multi-Agent Contextual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.15221","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intersectionzoo-eco-driving-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.15221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15221"}},"official":{"repos":["mit-wu-lab/IntersectionZoo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spa-bench-a-comprehensive-benchmark-for","slug":"spa-bench-a-comprehensive-benchmark-for","title":"SPA-Bench: A Comprehensive Benchmark for SmartPhone Agent Evaluation","date":"2024-10-19","arxiv_id":"2410.15164","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/spa-bench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.15164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15164"}},"official":{"repos":["ai-agents-2030/SPA-Bench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-deep-reinforcement-learning-for-1","slug":"benchmarking-deep-reinforcement-learning-for-1","title":"Benchmarking Deep Reinforcement Learning for Navigation in Denied Sensor Environments","date":"2024-10-18","arxiv_id":"2410.14616","repositories_listed":1,"syntology":null},{"url":"/paper/multichartqa-benchmarking-vision-language","slug":"multichartqa-benchmarking-vision-language","title":"MultiChartQA: Benchmarking Vision-Language Models on Multi-Chart Problems","date":"2024-10-18","arxiv_id":"2410.14179","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multichartqa-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.14179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14179"}},"official":{"repos":["zivenzhu/multi-chart-qa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ab-initio-nonparametric-variable-selection","slug":"ab-initio-nonparametric-variable-selection","title":"Ab Initio Nonparametric Variable Selection for Scalable Symbolic Regression with Large $p$","date":"2024-10-17","arxiv_id":"2410.13681","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ab-initio-nonparametric-variable-selection#ran","syntology_url":"https://syntology.ai/paper/2410.13681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13681"}},"official":{"repos":["mattsheng/PAN_SR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-transcriptomics-foundation","slug":"benchmarking-transcriptomics-foundation","title":"Benchmarking Transcriptomics Foundation Models for Perturbation Analysis : one PCA still rules them all","date":"2024-10-17","arxiv_id":"2410.13956","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-transcriptomics-foundation#ran","syntology_url":"https://syntology.ai/paper/2410.13956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13956"}},"official":{"repos":["valence-labs/Tx-Evaluation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-lingual-auto-evaluation-for-assessing","slug":"cross-lingual-auto-evaluation-for-assessing","title":"Cross-Lingual Auto Evaluation for Assessing Multilingual LLMs","date":"2024-10-17","arxiv_id":"2410.13394","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-lingual-auto-evaluation-for-assessing#ran","syntology_url":"https://syntology.ai/paper/2410.13394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13394"}},"official":{"repos":["ai4bharat/cia"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mapping-bias-in-vision-language-models","slug":"mapping-bias-in-vision-language-models","title":"debiaSAE: Benchmarking and Mitigating Vision-Language Model Bias","date":"2024-10-17","arxiv_id":"2410.13146","repositories_listed":1,"syntology":null},{"url":"/paper/orchid-a-chinese-debate-corpus-for-target","slug":"orchid-a-chinese-debate-corpus-for-target","title":"ORCHID: A Chinese Debate Corpus for Target-Independent Stance Detection and Argumentative Dialogue Summarization","date":"2024-10-17","arxiv_id":"2410.13667","repositories_listed":1,"syntology":null},{"url":"/paper/ucfe-a-user-centric-financial-expertise","slug":"ucfe-a-user-centric-financial-expertise","title":"UCFE: A User-Centric Financial Expertise Benchmark for Large Language Models","date":"2024-10-17","arxiv_id":"2410.14059","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-role-of-llms-in-multimodal","slug":"understanding-the-role-of-llms-in-multimodal","title":"Understanding the Role of LLMs in Multimodal Evaluation Benchmarks","date":"2024-10-16","arxiv_id":"2410.12329","repositories_listed":1,"syntology":null},{"url":"/paper/worldmedqa-v-a-multilingual-multimodal","slug":"worldmedqa-v-a-multilingual-multimodal","title":"WorldMedQA-V: a multilingual, multimodal medical examination dataset for multimodal language models evaluation","date":"2024-10-16","arxiv_id":"2410.12722","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-data-efficiency-in-d-ml-and","slug":"benchmarking-data-efficiency-in-d-ml-and","title":"Benchmarking Data Efficiency in $Δ$-ML and Multifidelity Models for Quantum Chemistry","date":"2024-10-15","arxiv_id":"2410.11391","repositories_listed":1,"syntology":null},{"url":"/paper/rclicks-realistic-click-simulation-for","slug":"rclicks-realistic-click-simulation-for","title":"RClicks: Realistic Click Simulation for Benchmarking Interactive Segmentation","date":"2024-10-15","arxiv_id":"2410.11722","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rclicks-realistic-click-simulation-for#ran","syntology_url":"https://syntology.ai/paper/2410.11722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11722"}},"official":{"repos":["emb-ai/rclicks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longmemeval-benchmarking-chat-assistants-on","slug":"longmemeval-benchmarking-chat-assistants-on","title":"LongMemEval: Benchmarking Chat Assistants on Long-Term Interactive Memory","date":"2024-10-14","arxiv_id":"2410.10813","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/longmemeval-benchmarking-chat-assistants-on#ran","syntology_url":"https://syntology.ai/paper/2410.10813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10813"}},"official":{"repos":["xiaowu0162/longmemeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-and-benchmarking-graph","slug":"revisiting-and-benchmarking-graph","title":"Revisiting and Benchmarking Graph Autoencoders: A Contrastive Learning Perspective","date":"2024-10-14","arxiv_id":"2410.10241","repositories_listed":1,"syntology":null},{"url":"/paper/sensorbench-benchmarking-llms-in-coding-based","slug":"sensorbench-benchmarking-llms-in-coding-based","title":"SensorBench: Benchmarking LLMs in Coding-Based Sensor Processing","date":"2024-10-14","arxiv_id":"2410.10741","repositories_listed":1,"syntology":null},{"url":"/paper/temporalbench-benchmarking-fine-grained","slug":"temporalbench-benchmarking-fine-grained","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","date":"2024-10-14","arxiv_id":"2410.10818","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporalbench-benchmarking-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.10818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10818"}},"official":{"repos":["mu-cai/TemporalBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-and-textual-graph-generation-via","slug":"dynamic-and-textual-graph-generation-via","title":"LLM-Based Multi-Agent Systems are Scalable Graph Generative Models","date":"2024-10-13","arxiv_id":"2410.09824","repositories_listed":1,"syntology":null},{"url":"/paper/loli-street-benchmarking-low-light-image","slug":"loli-street-benchmarking-low-light-image","title":"LoLI-Street: Benchmarking Low-Light Image Enhancement and Beyond","date":"2024-10-13","arxiv_id":"2410.09831","repositories_listed":1,"syntology":null},{"url":"/paper/rmb-comprehensively-benchmarking-reward","slug":"rmb-comprehensively-benchmarking-reward","title":"RMB: Comprehensively Benchmarking Reward Models in LLM Alignment","date":"2024-10-13","arxiv_id":"2410.09893","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rmb-comprehensively-benchmarking-reward#ran","syntology_url":"https://syntology.ai/paper/2410.09893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09893"}},"official":{"repos":["zhou-zoey/rmb-reward-model-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/fb-bench-a-fine-grained-multi-task-benchmark","slug":"fb-bench-a-fine-grained-multi-task-benchmark","title":"FB-Bench: A Fine-Grained Multi-Task Benchmark for Evaluating LLMs' Responsiveness to Human Feedback","date":"2024-10-12","arxiv_id":"2410.09412","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fb-bench-a-fine-grained-multi-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2410.09412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09412"}},"official":{"repos":["pku-baichuan-mlsystemlab/fb-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lexsumm-and-lext5-benchmarking-and-modeling","slug":"lexsumm-and-lext5-benchmarking-and-modeling","title":"LexSumm and LexT5: Benchmarking and Modeling Legal Summarization Tasks in English","date":"2024-10-12","arxiv_id":"2410.09527","repositories_listed":1,"syntology":null},{"url":"/paper/yesterday-s-news-benchmarking-multi","slug":"yesterday-s-news-benchmarking-multi","title":"Yesterday's News: Benchmarking Multi-Dimensional Out-of-Distribution Generalisation of Misinformation Detection Models","date":"2024-10-12","arxiv_id":"2410.18122","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-bidirectional-interaction-model","slug":"cross-modal-bidirectional-interaction-model","title":"Cross-Modal Bidirectional Interaction Model for Referring Remote Sensing Image Segmentation","date":"2024-10-11","arxiv_id":"2410.08613","repositories_listed":1,"syntology":null},{"url":"/paper/enterprise-benchmarks-for-large-language","slug":"enterprise-benchmarks-for-large-language","title":"Enterprise Benchmarks for Large Language Model Evaluation","date":"2024-10-11","arxiv_id":"2410.12857","repositories_listed":1,"syntology":null},{"url":"/paper/when-graph-meets-multimodal-benchmarking-on","slug":"when-graph-meets-multimodal-benchmarking-on","title":"When Graph meets Multimodal: Benchmarking on Multimodal Attributed Graphs Learning","date":"2024-10-11","arxiv_id":"2410.09132","repositories_listed":1,"syntology":null},{"url":"/paper/audio-explanation-synthesis-with-generative","slug":"audio-explanation-synthesis-with-generative","title":"Audio Explanation Synthesis with Generative Foundation Models","date":"2024-10-10","arxiv_id":"2410.07530","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-agentic-workflow-generation","slug":"benchmarking-agentic-workflow-generation","title":"Benchmarking Agentic Workflow Generation","date":"2024-10-10","arxiv_id":"2410.07869","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-agentic-workflow-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07869"}},"official":{"repos":["zjunlp/worfbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compl-ai-framework-a-technical-interpretation","slug":"compl-ai-framework-a-technical-interpretation","title":"COMPL-AI Framework: A Technical Interpretation and LLM Benchmarking Suite for the EU Artificial Intelligence Act","date":"2024-10-10","arxiv_id":"2410.07959","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compl-ai-framework-a-technical-interpretation#ran","syntology_url":"https://syntology.ai/paper/2410.07959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07959"}},"official":{"repos":["compl-ai/compl-ai"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/identifying-money-laundering-subgraphs-on-the","slug":"identifying-money-laundering-subgraphs-on-the","title":"Identifying Money Laundering Subgraphs on the Blockchain","date":"2024-10-10","arxiv_id":"2410.08394","repositories_listed":1,"syntology":null},{"url":"/paper/triage-ethical-benchmarking-of-ai-models","slug":"triage-ethical-benchmarking-of-ai-models","title":"TRIAGE: Ethical Benchmarking of AI Models Through Mass Casualty Simulations","date":"2024-10-10","arxiv_id":"2410.18991","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-data-heterogeneity-evaluation","slug":"benchmarking-data-heterogeneity-evaluation","title":"Benchmarking Data Heterogeneity Evaluation Approaches for Personalized Federated Learning","date":"2024-10-09","arxiv_id":"2410.07286","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-data-heterogeneity-evaluation#ran","syntology_url":"https://syntology.ai/paper/2410.07286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07286"}},"official":{"repos":["xiaoni-61/dh-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/quanda-an-interpretability-toolkit-for","slug":"quanda-an-interpretability-toolkit-for","title":"Quanda: An Interpretability Toolkit for Training Data Attribution Evaluation and Beyond","date":"2024-10-09","arxiv_id":"2410.07158","repositories_listed":1,"syntology":null},{"url":"/paper/towards-generalisable-time-series","slug":"towards-generalisable-time-series","title":"Towards Generalisable Time Series Understanding Across Domains","date":"2024-10-09","arxiv_id":"2410.07299","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/towards-generalisable-time-series#ran","syntology_url":"https://syntology.ai/paper/2410.07299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07299"}},"official":{"repos":["oetu/otis"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/turingq-benchmarking-ai-comprehension-in","slug":"turingq-benchmarking-ai-comprehension-in","title":"TuringQ: Benchmarking AI Comprehension in Theory of Computation","date":"2024-10-09","arxiv_id":"2410.06547","repositories_listed":1,"syntology":null},{"url":"/paper/entering-real-social-world-benchmarking-the","slug":"entering-real-social-world-benchmarking-the","title":"Entering Real Social World! Benchmarking the Social Intelligence of Large Language Models from a First-person Perspective","date":"2024-10-08","arxiv_id":"2410.06195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entering-real-social-world-benchmarking-the#ran","syntology_url":"https://syntology.ai/paper/2410.06195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06195"}},"official":{"repos":["gyhou123/egosocialarena"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fedgraph-a-research-library-and-benchmark-for","slug":"fedgraph-a-research-library-and-benchmark-for","title":"FedGraph: A Research Library and Benchmark for Federated Graph Learning","date":"2024-10-08","arxiv_id":"2410.06340","repositories_listed":1,"syntology":null},{"url":"/paper/qgym-scalable-simulation-and-benchmarking-of","slug":"qgym-scalable-simulation-and-benchmarking-of","title":"QGym: Scalable Simulation and Benchmarking of Queuing Network Controllers","date":"2024-10-08","arxiv_id":"2410.06170","repositories_listed":1,"syntology":null},{"url":"/paper/mibench-a-comprehensive-benchmark-for-model","slug":"mibench-a-comprehensive-benchmark-for-model","title":"MIBench: A Comprehensive Framework for Benchmarking Model Inversion Attack and Defense","date":"2024-10-07","arxiv_id":"2410.05159","repositories_listed":1,"syntology":null},{"url":"/paper/model-glue-democratized-llm-scaling-for-a","slug":"model-glue-democratized-llm-scaling-for-a","title":"Model-GLUE: Democratized LLM Scaling for A Large Model Zoo in the Wild","date":"2024-10-07","arxiv_id":"2410.05357","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":4,"n_ran_checked":6,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/model-glue-democratized-llm-scaling-for-a#ran","syntology_url":"https://syntology.ai/paper/2410.05357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05357"}},"official":{"repos":["model-glue/model-glue"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/named-clinical-entity-recognition-benchmark","slug":"named-clinical-entity-recognition-benchmark","title":"Named Clinical Entity Recognition Benchmark","date":"2024-10-07","arxiv_id":"2410.05046","repositories_listed":1,"syntology":null},{"url":"/paper/tunevlseg-prompt-tuning-benchmark-for-vision","slug":"tunevlseg-prompt-tuning-benchmark-for-vision","title":"TuneVLSeg: Prompt Tuning Benchmark for Vision-Language Segmentation Models","date":"2024-10-07","arxiv_id":"2410.05239","repositories_listed":1,"syntology":null},{"url":"/paper/adjusting-pretrained-backbones-for","slug":"adjusting-pretrained-backbones-for","title":"Adjusting Pretrained Backbones for Performativity","date":"2024-10-06","arxiv_id":"2410.04499","repositories_listed":1,"syntology":null},{"url":"/paper/cirrmri600-large-scale-mri-collection-and","slug":"cirrmri600-large-scale-mri-collection-and","title":"Large Scale MRI Collection and Segmentation of Cirrhotic Liver","date":"2024-10-06","arxiv_id":"2410.16296","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-dattri-a-library-for-efficient-data","slug":"texttt-dattri-a-library-for-efficient-data","title":"$\\texttt{dattri}$: A Library for Efficient Data Attribution","date":"2024-10-06","arxiv_id":"2410.04555","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/texttt-dattri-a-library-for-efficient-data#ran","syntology_url":"https://syntology.ai/paper/2410.04555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04555"}},"official":{"repos":["trais-lab/dattri"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-large-language-models-for-inverse","slug":"multimodal-large-language-models-for-inverse","title":"Multimodal Large Language Models for Inverse Molecular Design with Retrosynthetic Planning","date":"2024-10-05","arxiv_id":"2410.04223","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multimodal-large-language-models-for-inverse#ran","syntology_url":"https://syntology.ai/paper/2410.04223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04223"}},"official":{"repos":["liugangcode/Llamole"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tubench-benchmarking-large-vision-language","slug":"tubench-benchmarking-large-vision-language","title":"TUBench: Benchmarking Large Vision-Language Models on Trustworthiness with Unanswerable Questions","date":"2024-10-05","arxiv_id":"2410.04107","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-the-fidelity-and-utility-of","slug":"benchmarking-the-fidelity-and-utility-of","title":"Benchmarking the Fidelity and Utility of Synthetic Relational Data","date":"2024-10-04","arxiv_id":"2410.03411","repositories_listed":1,"syntology":null},{"url":"/paper/ebes-easy-benchmarking-for-event-sequences","slug":"ebes-easy-benchmarking-for-event-sequences","title":"EBES: Easy Benchmarking for Event Sequences","date":"2024-10-04","arxiv_id":"2410.03399","repositories_listed":1,"syntology":null},{"url":"/paper/lightning-uq-box-a-comprehensive-framework","slug":"lightning-uq-box-a-comprehensive-framework","title":"Lightning UQ Box: A Comprehensive Framework for Uncertainty Quantification in Deep Learning","date":"2024-10-04","arxiv_id":"2410.03390","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-benchmark-for-large-language-models","slug":"towards-a-benchmark-for-large-language-models","title":"Towards a Benchmark for Large Language Models for Business Process Management Tasks","date":"2024-10-04","arxiv_id":"2410.03255","repositories_listed":1,"syntology":null},{"url":"/paper/agent-security-bench-asb-formalizing-and","slug":"agent-security-bench-asb-formalizing-and","title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","date":"2024-10-03","arxiv_id":"2410.02644","repositories_listed":1,"syntology":{"n":13,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/agent-security-bench-asb-formalizing-and#ran","syntology_url":"https://syntology.ai/paper/2410.02644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02644"}},"official":{"repos":["agiresearch/asb"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/divscene-benchmarking-lvlms-for-object","slug":"divscene-benchmarking-lvlms-for-object","title":"DivScene: Benchmarking LVLMs for Object Navigation with Diverse Scenes and Objects","date":"2024-10-03","arxiv_id":"2410.02730","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divscene-benchmarking-lvlms-for-object#ran","syntology_url":"https://syntology.ai/paper/2410.02730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02730"}},"official":{"repos":["zhaowei-wang-nlp/divscene"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-pilot-characterize-and-optimize","slug":"llm-pilot-characterize-and-optimize","title":"LLM-Pilot: Characterize and Optimize Performance of your LLM Inference Services","date":"2024-10-03","arxiv_id":"2410.02425","repositories_listed":1,"syntology":null},{"url":"/paper/mantra-the-manifold-triangulations-assemblage","slug":"mantra-the-manifold-triangulations-assemblage","title":"MANTRA: The Manifold Triangulations Assemblage","date":"2024-10-03","arxiv_id":"2410.02392","repositories_listed":1,"syntology":{"n":18,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mantra-the-manifold-triangulations-assemblage#ran","syntology_url":"https://syntology.ai/paper/2410.02392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02392"}},"official":{"repos":["aidos-lab/MANTRA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/medqa-cs-benchmarking-large-language-models","slug":"medqa-cs-benchmarking-large-language-models","title":"MedQA-CS: Benchmarking Large Language Models Clinical Skills Using an AI-SCE Framework","date":"2024-10-02","arxiv_id":"2410.01553","repositories_listed":1,"syntology":null},{"url":"/paper/monica-benchmarking-on-long-tailed-medical","slug":"monica-benchmarking-on-long-tailed-medical","title":"MONICA: Benchmarking on Long-tailed Medical Image Classification","date":"2024-10-02","arxiv_id":"2410.02010","repositories_listed":1,"syntology":null},{"url":"/paper/shapiq-shapley-interactions-for-machine","slug":"shapiq-shapley-interactions-for-machine","title":"shapiq: Shapley Interactions for Machine Learning","date":"2024-10-02","arxiv_id":"2410.01649","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shapiq-shapley-interactions-for-machine#ran","syntology_url":"https://syntology.ai/paper/2410.01649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01649"}},"official":{"repos":["mmschlk/shapiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stringllm-understanding-the-string-processing","slug":"stringllm-understanding-the-string-processing","title":"StringLLM: Understanding the String Processing Capability of Large Language Models","date":"2024-10-02","arxiv_id":"2410.01208","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stringllm-understanding-the-string-processing#ran","syntology_url":"https://syntology.ai/paper/2410.01208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01208"}},"official":{"repos":["wxl-lxw/stringllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cxpmrg-bench-pre-training-and-benchmarking","slug":"cxpmrg-bench-pre-training-and-benchmarking","title":"CXPMRG-Bench: Pre-training and Benchmarking for X-ray Medical Report Generation on CheXpert Plus Dataset","date":"2024-10-01","arxiv_id":"2410.00379","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cxpmrg-bench-pre-training-and-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.00379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00379"}},"official":{"repos":["event-ahu/medical_image_analysis"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-prompts-dynamic-conversational","slug":"beyond-prompts-dynamic-conversational","title":"Beyond Prompts: Dynamic Conversational Benchmarking of Large Language Models","date":"2024-09-30","arxiv_id":"2409.20222","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-prompts-dynamic-conversational#ran","syntology_url":"https://syntology.ai/paper/2409.20222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20222"}},"official":{"repos":["GoodAI/goodai-ltm-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-quic-dynamics-a-large-scale-dataset","slug":"exploring-quic-dynamics-a-large-scale-dataset","title":"Exploring QUIC Dynamics: A Large-Scale Dataset for Encrypted Traffic Analysis","date":"2024-09-30","arxiv_id":"2410.03728","repositories_listed":1,"syntology":null},{"url":"/paper/immersepro-end-to-end-stereo-video-synthesis","slug":"immersepro-end-to-end-stereo-video-synthesis","title":"ImmersePro: End-to-End Stereo Video Synthesis Via Implicit Disparity Learning","date":"2024-09-30","arxiv_id":"2410.00262","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-graph-neural-networks-for-1","slug":"a-survey-on-graph-neural-networks-for-1","title":"A Survey on Graph Neural Networks for Remaining Useful Life Prediction: Methodologies, Evaluation and Future Trends","date":"2024-09-29","arxiv_id":"2409.19629","repositories_listed":1,"syntology":null},{"url":"/paper/constrained-reinforcement-learning-for-safe","slug":"constrained-reinforcement-learning-for-safe","title":"Constrained Reinforcement Learning for Safe Heat Pump Control","date":"2024-09-29","arxiv_id":"2409.19716","repositories_listed":1,"syntology":null},{"url":"/paper/arlbench-flexible-and-efficient-benchmarking","slug":"arlbench-flexible-and-efficient-benchmarking","title":"ARLBench: Flexible and Efficient Benchmarking for Hyperparameter Optimization in Reinforcement Learning","date":"2024-09-27","arxiv_id":"2409.18827","repositories_listed":1,"syntology":null},{"url":"/paper/constructing-confidence-intervals-for-the-1","slug":"constructing-confidence-intervals-for-the-1","title":"Constructing Confidence Intervals for 'the' Generalization Error -- a Comprehensive Benchmark Study","date":"2024-09-27","arxiv_id":"2409.18836","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-graph-conformal-prediction","slug":"benchmarking-graph-conformal-prediction","title":"Conformal Prediction: A Theoretical Note and Benchmarking Transductive Node Classification in Graphs","date":"2024-09-26","arxiv_id":"2409.18332","repositories_listed":1,"syntology":null},{"url":"/paper/malpolon-a-framework-for-deep-species","slug":"malpolon-a-framework-for-deep-species","title":"MALPOLON: A Framework for Deep Species Distribution Modeling","date":"2024-09-26","arxiv_id":"2409.18102","repositories_listed":1,"syntology":null},{"url":"/paper/the-elephant-in-the-room-towards-a-reliable","slug":"the-elephant-in-the-room-towards-a-reliable","title":"The Elephant in the Room: Towards A Reliable Time-Series Anomaly Detection Benchmark","date":"2024-09-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-domain-generalization-algorithms","slug":"benchmarking-domain-generalization-algorithms","title":"Benchmarking Domain Generalization Algorithms in Computational Pathology","date":"2024-09-25","arxiv_id":"2409.17063","repositories_listed":1,"syntology":null},{"url":"/paper/hazespace2m-a-dataset-for-haze-aware-single","slug":"hazespace2m-a-dataset-for-haze-aware-single","title":"HazeSpace2M: A Dataset for Haze Aware Single Image Dehazing","date":"2024-09-25","arxiv_id":"2409.17432","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-robustness-of-endoscopic-depth","slug":"benchmarking-robustness-of-endoscopic-depth","title":"Benchmarking Robustness of Endoscopic Depth Estimation with Synthetically Corrupted Data","date":"2024-09-24","arxiv_id":"2409.16063","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-risk-of-retrieval-augmented","slug":"controlling-risk-of-retrieval-augmented","title":"Controlling Risk of Retrieval-augmented Generation: A Counterfactual Prompting Framework","date":"2024-09-24","arxiv_id":"2409.16146","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/controlling-risk-of-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2409.16146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16146"}},"official":{"repos":["ict-bigdatalab/rc-rag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ducho-meets-elliot-large-scale-benchmarks-for-1","slug":"ducho-meets-elliot-large-scale-benchmarks-for-1","title":"Ducho meets Elliot: Large-scale Benchmarks for Multimodal Recommendation","date":"2024-09-24","arxiv_id":"2409.15857","repositories_listed":1,"syntology":null},{"url":"/paper/gsplatloc-grounding-keypoint-descriptors-into","slug":"gsplatloc-grounding-keypoint-descriptors-into","title":"GSplatLoc: Grounding Keypoint Descriptors into 3D Gaussian Splatting for Improved Visual Localization","date":"2024-09-24","arxiv_id":"2409.16502","repositories_listed":1,"syntology":null},{"url":"/paper/small-language-models-survey-measurements-and","slug":"small-language-models-survey-measurements-and","title":"Small Language Models: Survey, Measurements, and Insights","date":"2024-09-24","arxiv_id":"2409.15790","repositories_listed":1,"syntology":null},{"url":"/paper/alphazip-neural-network-enhanced-lossless","slug":"alphazip-neural-network-enhanced-lossless","title":"AlphaZip: Neural Network-Enhanced Lossless Text Compression","date":"2024-09-23","arxiv_id":"2409.15046","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-healthcare-llms-through-retrieved","slug":"boosting-healthcare-llms-through-retrieved","title":"Boosting Healthcare LLMs Through Retrieved Context","date":"2024-09-23","arxiv_id":"2409.15127","repositories_listed":1,"syntology":null},{"url":"/paper/rmcbench-benchmarking-large-language-models","slug":"rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","arxiv_id":"2409.15154","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rmcbench-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2409.15154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15154"}},"official":{"repos":["qing-yuan233/RMCBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/style-over-substance-failure-modes-of-llm","slug":"style-over-substance-failure-modes-of-llm","title":"Style Outweighs Substance: Failure Modes of LLM Judges in Alignment Benchmarking","date":"2024-09-23","arxiv_id":"2409.15268","repositories_listed":1,"syntology":null},{"url":"/paper/towards-ground-truth-free-evaluation-of-any","slug":"towards-ground-truth-free-evaluation-of-any","title":"Towards Ground-truth-free Evaluation of Any Segmentation in Medical Images","date":"2024-09-23","arxiv_id":"2409.14874","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-impact-of-hard-samples-on","slug":"investigating-the-impact-of-hard-samples-on","title":"Investigating the Impact of Hard Samples on Accuracy Reveals In-class Data Imbalance","date":"2024-09-22","arxiv_id":"2409.14401","repositories_listed":1,"syntology":null},{"url":"/paper/margin-bounded-confidence-scores-for-out-of","slug":"margin-bounded-confidence-scores-for-out-of","title":"Margin-bounded Confidence Scores for Out-of-Distribution Detection","date":"2024-09-22","arxiv_id":"2410.07185","repositories_listed":1,"syntology":null},{"url":"/paper/2409-14037","slug":"2409-14037","title":"Can LLMs replace Neil deGrasse Tyson? Evaluating the Reliability of LLMs as Science Communicators","date":"2024-09-21","arxiv_id":"2409.14037","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-multimodal-benchmarks-in-the-era","slug":"a-survey-on-multimodal-benchmarks-in-the-era","title":"A Survey on Multimodal Benchmarks: In the Era of Large AI Models","date":"2024-09-21","arxiv_id":"2409.18142","repositories_listed":1,"syntology":null},{"url":"/paper/congra-benchmarking-automatic-conflict","slug":"congra-benchmarking-automatic-conflict","title":"CONGRA: Benchmarking Automatic Conflict Resolution","date":"2024-09-21","arxiv_id":"2409.14121","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-effective-model-extraction","slug":"efficient-and-effective-model-extraction","title":"Efficient and Effective Model Extraction","date":"2024-09-21","arxiv_id":"2409.14122","repositories_listed":1,"syntology":null},{"url":"/paper/present-and-future-generalization-of","slug":"present-and-future-generalization-of","title":"Present and Future Generalization of Synthetic Image Detectors","date":"2024-09-21","arxiv_id":"2409.14128","repositories_listed":1,"syntology":null},{"url":"/paper/stop-benchmarking-large-language-models-with","slug":"stop-benchmarking-large-language-models-with","title":"STOP! Benchmarking Large Language Models with Sensitivity Testing on Offensive Progressions","date":"2024-09-20","arxiv_id":"2409.13843","repositories_listed":1,"syntology":null},{"url":"/paper/yesbut-a-high-quality-annotated-multimodal","slug":"yesbut-a-high-quality-annotated-multimodal","title":"YesBut: A High-Quality Annotated Multimodal Dataset for evaluating Satire Comprehension capability of Vision-Language Models","date":"2024-09-20","arxiv_id":"2409.13592","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-performance-tracking-leveraging","slug":"efficient-performance-tracking-leveraging","title":"Efficient Performance Tracking: Leveraging Large Language Models for Automated Construction of Scientific Leaderboards","date":"2024-09-19","arxiv_id":"2409.12656","repositories_listed":1,"syntology":null},{"url":"/paper/asr-benchmarking-need-for-a-more","slug":"asr-benchmarking-need-for-a-more","title":"ASR Benchmarking: Need for a More Representative Conversational Dataset","date":"2024-09-18","arxiv_id":"2409.12042","repositories_listed":1,"syntology":null},{"url":"/paper/hard-label-cryptanalytic-extraction-of-neural","slug":"hard-label-cryptanalytic-extraction-of-neural","title":"Hard-Label Cryptanalytic Extraction of Neural Network Models","date":"2024-09-18","arxiv_id":"2409.11646","repositories_listed":1,"syntology":null},{"url":"/paper/paraphrasus-a-comprehensive-benchmark-for","slug":"paraphrasus-a-comprehensive-benchmark-for","title":"PARAPHRASUS : A Comprehensive Benchmark for Evaluating Paraphrase Detection Models","date":"2024-09-18","arxiv_id":"2409.12060","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/paraphrasus-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2409.12060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12060"}},"official":{"repos":["impresso/paraphrasus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"efc7b5b02000b335b239d9db0bf3c5a3ba2eea4ed6625b12f456fbf46f91cc7e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}