{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/2","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":56,"rows_per_page":100,"rows":[101,200],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking","next":"/task/benchmarking/papers/3","papers":[{"url":"/paper/better-than-classical-the-subtle-art-of","slug":"better-than-classical-the-subtle-art-of","title":"Better than classical? The subtle art of benchmarking quantum machine learning models","date":"2024-03-11","arxiv_id":"2403.07059","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-than-classical-the-subtle-art-of#ran","syntology_url":"https://syntology.ai/paper/2403.07059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07059"}},"official":{"repos":["xanaduai/qml-benchmarks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/exponentially-faster-language-modelling","slug":"exponentially-faster-language-modelling","title":"Exponentially Faster Language Modelling","date":"2023-11-15","arxiv_id":"2311.10770","repositories_listed":3,"syntology":null},{"url":"/paper/waterbench-towards-holistic-evaluation-of","slug":"waterbench-towards-holistic-evaluation-of","title":"WaterBench: Towards Holistic Evaluation of Watermarks for Large Language Models","date":"2023-11-13","arxiv_id":"2311.07138","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/waterbench-towards-holistic-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2311.07138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07138"}},"official":{"repos":["THU-KEG/WaterBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/benchmarking-the-abilities-of-large-language","slug":"benchmarking-the-abilities-of-large-language","title":"Benchmarking the Abilities of Large Language Models for RDF Knowledge Graph Creation and Comprehension: How Well Do LLMs Speak Turtle?","date":"2023-09-29","arxiv_id":"2309.17122","repositories_listed":3,"syntology":null},{"url":"/paper/revisiting-neural-program-smoothing-for","slug":"revisiting-neural-program-smoothing-for","title":"Revisiting Neural Program Smoothing for Fuzzing","date":"2023-09-28","arxiv_id":"2309.16618","repositories_listed":3,"syntology":null},{"url":"/paper/matbench-discovery-an-evaluation-framework","slug":"matbench-discovery-an-evaluation-framework","title":"Matbench Discovery -- A framework to evaluate machine learning crystal stability predictions","date":"2023-08-28","arxiv_id":"2308.14920","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/matbench-discovery-an-evaluation-framework#ran","syntology_url":"https://syntology.ai/paper/2308.14920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14920"}},"official":{"repos":["janosh/matbench-discovery","janosh/pymatviz"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seed-bench-benchmarking-multimodal-llms-with","slug":"seed-bench-benchmarking-multimodal-llms-with","title":"SEED-Bench: Benchmarking Multimodal LLMs with Generative Comprehension","date":"2023-07-30","arxiv_id":"2307.16125","repositories_listed":3,"syntology":null},{"url":"/paper/limits-of-machine-learning-for-automatic","slug":"limits-of-machine-learning-for-automatic","title":"Uncovering the Limits of Machine Learning for Automatic Vulnerability Detection","date":"2023-06-28","arxiv_id":"2306.17193","repositories_listed":3,"syntology":null},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","slug":"datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/datasets-and-benchmarks-for-offline-safe#ran","syntology_url":"https://syntology.ai/paper/2306.09303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09303"}},"official":{"repos":["liuzuxin/dsrl","liuzuxin/fsrl","liuzuxin/osrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/popgym-benchmarking-partially-observable","slug":"popgym-benchmarking-partially-observable","title":"POPGym: Benchmarking Partially Observable Reinforcement Learning","date":"2023-03-03","arxiv_id":"2303.01859","repositories_listed":3,"syntology":null},{"url":"/paper/aer-auto-encoder-with-regression-for-time","slug":"aer-auto-encoder-with-regression-for-time","title":"AER: Auto-Encoder with Regression for Time Series Anomaly Detection","date":"2022-12-27","arxiv_id":"2212.13558","repositories_listed":3,"syntology":null},{"url":"/paper/a-comprehensive-benchmark-for-covid-19","slug":"a-comprehensive-benchmark-for-covid-19","title":"A Comprehensive Benchmark for COVID-19 Predictive Modeling Using Electronic Health Records in Intensive Care","date":"2022-09-16","arxiv_id":"2209.07805","repositories_listed":3,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-comprehensive-benchmark-for-covid-19#ran","syntology_url":"https://syntology.ai/paper/2209.07805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.07805"}},"official":{"repos":["yhzhu99/covid-ehr-benchmarks","yhzhu99/pyehr","yhzhu99/pyehr-playground"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-performance-of-long-document","slug":"understanding-performance-of-long-document","title":"Understanding Performance of Long-Document Ranking Models through Comprehensive Evaluation and Leaderboarding","date":"2022-07-04","arxiv_id":"2207.01262","repositories_listed":3,"syntology":null},{"url":"/paper/beyond-neural-scaling-laws-beating-power-law","slug":"beyond-neural-scaling-laws-beating-power-law","title":"Beyond neural scaling laws: beating power law scaling via data pruning","date":"2022-06-29","arxiv_id":"2206.14486","repositories_listed":3,"syntology":null},{"url":"/paper/benchopt-reproducible-efficient-and","slug":"benchopt-reproducible-efficient-and","title":"Benchopt: Reproducible, efficient and collaborative optimization benchmarks","date":"2022-06-27","arxiv_id":"2206.13424","repositories_listed":3,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchopt-reproducible-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2206.13424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.13424"}},"official":{"repos":["benchopt/benchopt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/parameter-efficient-fine-tuning-for-vision","slug":"parameter-efficient-fine-tuning-for-vision","title":"Parameter-efficient Model Adaptation for Vision Transformers","date":"2022-03-29","arxiv_id":"2203.16329","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/parameter-efficient-fine-tuning-for-vision#ran","syntology_url":"https://syntology.ai/paper/2203.16329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16329"}},"official":{"repos":["eric-ai-lab/pevit"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mutual-information-based-few-shot","slug":"mutual-information-based-few-shot","title":"Mutual-Information Based Few-Shot Classification","date":"2021-06-23","arxiv_id":"2106.12252","repositories_listed":3,"syntology":null},{"url":"/paper/beir-a-heterogenous-benchmark-for-zero-shot","slug":"beir-a-heterogenous-benchmark-for-zero-shot","title":"BEIR: A Heterogenous Benchmark for Zero-shot Evaluation of Information Retrieval Models","date":"2021-04-17","arxiv_id":"2104.08663","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beir-a-heterogenous-benchmark-for-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2104.08663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08663"}},"official":{"repos":["UKPLab/beir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarks-for-deep-off-policy-evaluation-1","slug":"benchmarks-for-deep-off-policy-evaluation-1","title":"Benchmarks for Deep Off-Policy Evaluation","date":"2021-03-30","arxiv_id":"2103.16596","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarks-for-deep-off-policy-evaluation-1#ran","syntology_url":"https://syntology.ai/paper/2103.16596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.16596"}},"official":{"repos":["google-research/deep_ope"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/trafficqa-a-question-answering-benchmark-and","slug":"trafficqa-a-question-answering-benchmark-and","title":"SUTD-TrafficQA: A Question Answering Benchmark and an Efficient Network for Video Reasoning over Traffic Events","date":"2021-03-29","arxiv_id":"2103.15538","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":1,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trafficqa-a-question-answering-benchmark-and#ran","syntology_url":"https://syntology.ai/paper/2103.15538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.15538"}},"official":{"repos":["SUTDCV/SUTD-TrafficQA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-to-fly-a-gym-environment-with","slug":"learning-to-fly-a-gym-environment-with","title":"Learning to Fly -- a Gym Environment with PyBullet Physics for Reinforcement Learning of Multi-agent Quadcopter Control","date":"2021-03-03","arxiv_id":"2103.02142","repositories_listed":3,"syntology":null},{"url":"/paper/catching-out-of-context-misinformation-with","slug":"catching-out-of-context-misinformation-with","title":"COSMOS: Catching Out-of-Context Misinformation with Self-Supervised Learning","date":"2021-01-15","arxiv_id":"2101.06278","repositories_listed":3,"syntology":null},{"url":"/paper/pmlb-v1-0-an-open-source-dataset-collection","slug":"pmlb-v1-0-an-open-source-dataset-collection","title":"PMLB v1.0: An open source dataset collection for benchmarking machine learning methods","date":"2020-11-30","arxiv_id":"2012.00058","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/pmlb-v1-0-an-open-source-dataset-collection#ran","syntology_url":"https://syntology.ai/paper/2012.00058","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.00058"}},"official":{"repos":["EpistasisLab/pmlb","EpistasisLab/pmlbr","EpistasisLab/pmlb-manuscript"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/collective-knowledge-organizing-research","slug":"collective-knowledge-organizing-research","title":"Collective Knowledge: organizing research projects as a database of reusable components and portable workflows with common APIs","date":"2020-11-02","arxiv_id":"2011.01149","repositories_listed":3,"syntology":null},{"url":"/paper/indonlu-benchmark-and-resources-for","slug":"indonlu-benchmark-and-resources-for","title":"IndoNLU: Benchmark and Resources for Evaluating Indonesian Natural Language Understanding","date":"2020-09-11","arxiv_id":"2009.05387","repositories_listed":3,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/indonlu-benchmark-and-resources-for#ran","syntology_url":"https://syntology.ai/paper/2009.05387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.05387"}},"official":{"repos":["indobenchmark/indonlu"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gama-a-general-automated-machine-learning","slug":"gama-a-general-automated-machine-learning","title":"GAMA: a General Automated Machine learning Assistant","date":"2020-07-09","arxiv_id":"2007.04911","repositories_listed":3,"syntology":{"n":17,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gama-a-general-automated-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2007.04911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.04911"}},"official":{"repos":["PGijsbers/gama"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/iohanalyzer-performance-analysis-for","slug":"iohanalyzer-performance-analysis-for","title":"IOHanalyzer: Detailed Performance Analyses for Iterative Optimization Heuristics","date":"2020-07-08","arxiv_id":"2007.03953","repositories_listed":3,"syntology":null},{"url":"/paper/introducing-the-voiceprivacy-initiative","slug":"introducing-the-voiceprivacy-initiative","title":"Introducing the VoicePrivacy Initiative","date":"2020-05-04","arxiv_id":"2005.01387","repositories_listed":3,"syntology":null},{"url":"/paper/empirical-study-of-off-policy-policy","slug":"empirical-study-of-off-policy-policy","title":"Empirical Study of Off-Policy Policy Evaluation for Reinforcement Learning","date":"2019-11-15","arxiv_id":"1911.06854","repositories_listed":3,"syntology":null},{"url":"/paper/word-level-deep-sign-language-recognition","slug":"word-level-deep-sign-language-recognition","title":"Word-level Deep Sign Language Recognition from Video: A New Large-scale Dataset and Methods Comparison","date":"2019-10-24","arxiv_id":"1910.11006","repositories_listed":3,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/word-level-deep-sign-language-recognition#ran","syntology_url":"https://syntology.ai/paper/1910.11006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.11006"}},"official":{"repos":["dxli94/WLASL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/190600270","slug":"190600270","title":"Natural Image Noise Dataset","date":"2019-06-01","arxiv_id":"1906.00270","repositories_listed":3,"syntology":null},{"url":"/paper/simitate-a-hybrid-imitation-learning","slug":"simitate-a-hybrid-imitation-learning","title":"Simitate: A Hybrid Imitation Learning Benchmark","date":"2019-05-15","arxiv_id":"1905.06002","repositories_listed":3,"syntology":null},{"url":"/paper/2017-robotic-instrument-segmentation","slug":"2017-robotic-instrument-segmentation","title":"2017 Robotic Instrument Segmentation Challenge","date":"2019-02-18","arxiv_id":"1902.06426","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2017-robotic-instrument-segmentation#ran","syntology_url":"https://syntology.ai/paper/1902.06426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.06426"}},"official":{"repos":["duggalrahul/MICCAI17_EndoVis_RoboSeg","ternaus/robot-surgery-segmentation"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evalai-towards-better-evaluation-systems-for","slug":"evalai-towards-better-evaluation-systems-for","title":"EvalAI: Towards Better Evaluation Systems for AI Agents","date":"2019-02-10","arxiv_id":"1902.03570","repositories_listed":3,"syntology":null},{"url":"/paper/moment-matching-for-multi-source-domain","slug":"moment-matching-for-multi-source-domain","title":"Moment Matching for Multi-Source Domain Adaptation","date":"2018-12-04","arxiv_id":"1812.01754","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/moment-matching-for-multi-source-domain#ran","syntology_url":"https://syntology.ai/paper/1812.01754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.01754"}},"official":null}},{"url":"/paper/molecular-sets-moses-a-benchmarking-platform","slug":"molecular-sets-moses-a-benchmarking-platform","title":"Molecular Sets (MOSES): A Benchmarking Platform for Molecular Generation Models","date":"2018-11-29","arxiv_id":"1811.12823","repositories_listed":3,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/molecular-sets-moses-a-benchmarking-platform#ran","syntology_url":"https://syntology.ai/paper/1811.12823","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.12823"}},"official":{"repos":["molecularsets/moses"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/guacamol-benchmarking-models-for-de-novo","slug":"guacamol-benchmarking-models-for-de-novo","title":"GuacaMol: Benchmarking Models for De Novo Molecular Design","date":"2018-11-22","arxiv_id":"1811.09621","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/guacamol-benchmarking-models-for-de-novo#ran","syntology_url":"https://syntology.ai/paper/1811.09621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.09621"}},"official":{"repos":["benevolentAI/guacamol","benevolentAI/guacamol_baselines"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sequence-aware-recommender-systems","slug":"sequence-aware-recommender-systems","title":"Sequence-Aware Recommender Systems","date":"2018-02-23","arxiv_id":"1802.08452","repositories_listed":3,"syntology":null},{"url":"/paper/forecasting-across-time-series-databases","slug":"forecasting-across-time-series-databases","title":"Forecasting Across Time Series Databases using Recurrent Neural Networks on Groups of Similar Series: A Clustering Approach","date":"2017-10-09","arxiv_id":"1710.03222","repositories_listed":3,"syntology":null},{"url":"/paper/introducing-slambench-a-performance-and","slug":"introducing-slambench-a-performance-and","title":"Introducing SLAMBench, a performance and accuracy benchmarking methodology for SLAM","date":"2014-10-08","arxiv_id":"1410.2167","repositories_listed":3,"syntology":null},{"url":"/paper/the-arcade-learning-environment-an-evaluation","slug":"the-arcade-learning-environment-an-evaluation","title":"The Arcade Learning Environment: An Evaluation Platform for General Agents","date":"2012-07-19","arxiv_id":"1207.4708","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-arcade-learning-environment-an-evaluation#ran","syntology_url":"https://syntology.ai/paper/1207.4708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1207.4708"}},"official":null}},{"url":"/paper/tab-unified-benchmarking-of-time-series","slug":"tab-unified-benchmarking-of-time-series","title":"TAB: Unified Benchmarking of Time Series Anomaly Detection Methods","date":"2025-06-22","arxiv_id":"2506.18046","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tab-unified-benchmarking-of-time-series#ran","syntology_url":"https://syntology.ai/paper/2506.18046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18046"}},"official":{"repos":["decisionintelligence/tab"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/universal-music-representations-evaluating","slug":"universal-music-representations-evaluating","title":"Universal Music Representations? Evaluating Foundation Models on World Music Corpora","date":"2025-06-20","arxiv_id":"2506.17055","repositories_listed":2,"syntology":null},{"url":"/paper/mca-bench-a-multimodal-benchmark-for","slug":"mca-bench-a-multimodal-benchmark-for","title":"MCA-Bench: A Multimodal Benchmark for Evaluating CAPTCHA Robustness Against VLM-based Attacks","date":"2025-06-06","arxiv_id":"2506.05982","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mca-bench-a-multimodal-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2506.05982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05982"}},"official":{"repos":["noheadwuzonglin/mca-bench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/llamea-bo-a-large-language-model-evolutionary","slug":"llamea-bo-a-large-language-model-evolutionary","title":"LLaMEA-BO: A Large Language Model Evolutionary Algorithm for Automatically Generating Bayesian Optimization Algorithms","date":"2025-05-27","arxiv_id":"2505.21034","repositories_listed":2,"syntology":null},{"url":"/paper/finlora-benchmarking-lora-methods-for-fine","slug":"finlora-benchmarking-lora-methods-for-fine","title":"FinLoRA: Benchmarking LoRA Methods for Fine-Tuning LLMs on Financial Datasets","date":"2025-05-26","arxiv_id":"2505.19819","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finlora-benchmarking-lora-methods-for-fine#ran","syntology_url":"https://syntology.ai/paper/2505.19819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19819"}},"official":{"repos":["open-finance-lab/finlora"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-rethinking-knowledge-editing","slug":"benchmarking-and-rethinking-knowledge-editing","title":"Benchmarking and Rethinking Knowledge Editing for Large Language Models","date":"2025-05-24","arxiv_id":"2505.18690","repositories_listed":2,"syntology":null},{"url":"/paper/generative-models-for-fast-simulation-of","slug":"generative-models-for-fast-simulation-of","title":"Generative Models for Fast Simulation of Cherenkov Detectors at the Electron-Ion Collider","date":"2025-04-26","arxiv_id":"2504.19042","repositories_listed":2,"syntology":null},{"url":"/paper/hypobench-towards-systematic-and-principled","slug":"hypobench-towards-systematic-and-principled","title":"HypoBench: Towards Systematic and Principled Benchmarking for Hypothesis Generation","date":"2025-04-15","arxiv_id":"2504.11524","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypobench-towards-systematic-and-principled#ran","syntology_url":"https://syntology.ai/paper/2504.11524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11524"}},"official":null}},{"url":"/paper/real-benchmarking-autonomous-agents-on","slug":"real-benchmarking-autonomous-agents-on","title":"REAL: Benchmarking Autonomous Agents on Deterministic Simulations of Real Websites","date":"2025-04-15","arxiv_id":"2504.11543","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-benchmarking-autonomous-agents-on#ran","syntology_url":"https://syntology.ai/paper/2504.11543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11543"}},"official":{"repos":["agi-inc/real","agi-inc/agisdk"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lmm4lmm-benchmarking-and-evaluating-large","slug":"lmm4lmm-benchmarking-and-evaluating-large","title":"LMM4LMM: Benchmarking and Evaluating Large-multimodal Image Generation with LMMs","date":"2025-04-11","arxiv_id":"2504.08358","repositories_listed":2,"syntology":null},{"url":"/paper/icebench-a-benchmark-for-deep-learning-based","slug":"icebench-a-benchmark-for-deep-learning-based","title":"IceBench: A Benchmark for Deep Learning based Sea Ice Type Classification","date":"2025-03-22","arxiv_id":"2503.17877","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-dynamic-slo-compliance-in","slug":"benchmarking-dynamic-slo-compliance-in","title":"Benchmarking Dynamic SLO Compliance in Distributed Computing Continuum Systems","date":"2025-03-05","arxiv_id":"2503.03274","repositories_listed":2,"syntology":null},{"url":"/paper/milic-eval-benchmarking-multilingual-llms-for","slug":"milic-eval-benchmarking-multilingual-llms-for","title":"MiLiC-Eval: Benchmarking Multilingual LLMs for China's Minority Languages","date":"2025-03-03","arxiv_id":"2503.01150","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-retrieval-augmented-generation-1","slug":"benchmarking-retrieval-augmented-generation-1","title":"Benchmarking Retrieval-Augmented Generation in Multi-Modal Contexts","date":"2025-02-24","arxiv_id":"2502.17297","repositories_listed":2,"syntology":null},{"url":"/paper/accelerating-data-processing-and-benchmarking","slug":"accelerating-data-processing-and-benchmarking","title":"Accelerating Data Processing and Benchmarking of AI Models for Pathology","date":"2025-02-10","arxiv_id":"2502.06750","repositories_listed":2,"syntology":null},{"url":"/paper/itbench-evaluating-ai-agents-across-diverse","slug":"itbench-evaluating-ai-agents-across-diverse","title":"ITBench: Evaluating AI Agents across Diverse Real-World IT Automation Tasks","date":"2025-02-07","arxiv_id":"2502.05352","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/itbench-evaluating-ai-agents-across-diverse#ran","syntology_url":"https://syntology.ai/paper/2502.05352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05352"}},"official":{"repos":["IBM/itbench-sample-scenarios","IBM/itbench-sre-agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-perturbation-based-explanation","slug":"improving-the-perturbation-based-explanation","title":"Improving the Perturbation-Based Explanation of Deepfake Detectors Through the Use of Adversarially-Generated Samples","date":"2025-02-06","arxiv_id":"2502.03957","repositories_listed":2,"syntology":null},{"url":"/paper/molecular-driven-foundation-model-for","slug":"molecular-driven-foundation-model-for","title":"Molecular-driven Foundation Model for Oncologic Pathology","date":"2025-01-28","arxiv_id":"2501.16652","repositories_listed":2,"syntology":null},{"url":"/paper/evorl-a-gpu-accelerated-framework-for","slug":"evorl-a-gpu-accelerated-framework-for","title":"EvoRL: A GPU-accelerated Framework for Evolutionary Reinforcement Learning","date":"2025-01-25","arxiv_id":"2501.15129","repositories_listed":2,"syntology":null},{"url":"/paper/webwalker-benchmarking-llms-in-web-traversal","slug":"webwalker-benchmarking-llms-in-web-traversal","title":"WebWalker: Benchmarking LLMs in Web Traversal","date":"2025-01-13","arxiv_id":"2501.07572","repositories_listed":2,"syntology":null},{"url":"/paper/theagentcompany-benchmarking-llm-agents-on","slug":"theagentcompany-benchmarking-llm-agents-on","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","date":"2024-12-18","arxiv_id":"2412.14161","repositories_listed":2,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/theagentcompany-benchmarking-llm-agents-on#ran","syntology_url":"https://syntology.ai/paper/2412.14161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14161"}},"official":{"repos":["theagentcompany/experiments","theagentcompany/theagentcompany"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ad-llm-benchmarking-large-language-models-for","slug":"ad-llm-benchmarking-large-language-models-for","title":"AD-LLM: Benchmarking Large Language Models for Anomaly Detection","date":"2024-12-15","arxiv_id":"2412.11142","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ad-llm-benchmarking-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2412.11142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11142"}},"official":{"repos":["usc-fortis/ad-llm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/conqret-benchmarking-fine-grained-evaluation","slug":"conqret-benchmarking-fine-grained-evaluation","title":"ConQRet: Benchmarking Fine-Grained Evaluation of Retrieval Augmented Argumentation with LLM Judges","date":"2024-12-06","arxiv_id":"2412.05206","repositories_listed":2,"syntology":null},{"url":"/paper/magnetic-resonance-imaging-feature-based","slug":"magnetic-resonance-imaging-feature-based","title":"Magnetic Resonance Imaging Feature-Based Subtyping and Model Ensemble for Enhanced Brain Tumor Segmentation","date":"2024-12-05","arxiv_id":"2412.04094","repositories_listed":2,"syntology":null},{"url":"/paper/injecguard-benchmarking-and-mitigating-over","slug":"injecguard-benchmarking-and-mitigating-over","title":"InjecGuard: Benchmarking and Mitigating Over-defense in Prompt Injection Guardrail Models","date":"2024-10-30","arxiv_id":"2410.22770","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/injecguard-benchmarking-and-mitigating-over#ran","syntology_url":"https://syntology.ai/paper/2410.22770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22770"}},"official":{"repos":["leolee99/injecguard","safolab-wisc/injecguard"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mlperf-power-benchmarking-the-energy","slug":"mlperf-power-benchmarking-the-energy","title":"MLPerf Power: Benchmarking the Energy Efficiency of Machine Learning Systems from Microwatts to Megawatts for Sustainable AI","date":"2024-10-15","arxiv_id":"2410.12032","repositories_listed":2,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/mlperf-power-benchmarking-the-energy#ran","syntology_url":"https://syntology.ai/paper/2410.12032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12032"}},"official":null}},{"url":"/paper/autopenbench-benchmarking-generative-agents","slug":"autopenbench-benchmarking-generative-agents","title":"AutoPenBench: Benchmarking Generative Agents for Penetration Testing","date":"2024-10-04","arxiv_id":"2410.03225","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/autopenbench-benchmarking-generative-agents#ran","syntology_url":"https://syntology.ai/paper/2410.03225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03225"}},"official":{"repos":["lucagioacchini/auto-pen-bench","lucagioacchini/genai-pentest-paper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/assessing-sparql-capabilities-of-large","slug":"assessing-sparql-capabilities-of-large","title":"Assessing SPARQL capabilities of Large Language Models","date":"2024-09-09","arxiv_id":"2409.05925","repositories_listed":2,"syntology":null},{"url":"/paper/llm-detectors-still-fall-short-of-real-world","slug":"llm-detectors-still-fall-short-of-real-world","title":"LLM Detectors Still Fall Short of Real World: Case of LLM-Generated Short News-Like Posts","date":"2024-09-05","arxiv_id":"2409.03291","repositories_listed":2,"syntology":null},{"url":"/paper/2409-13694","slug":"2409-13694","title":"Multi-Source Knowledge Pruning for Retrieval-Augmented Generation: A Benchmark and Empirical Study","date":"2024-09-03","arxiv_id":"2409.13694","repositories_listed":2,"syntology":null},{"url":"/paper/spinning-the-golden-thread-benchmarking-long","slug":"spinning-the-golden-thread-benchmarking-long","title":"LongGenBench: Benchmarking Long-Form Generation in Long Context LLMs","date":"2024-09-03","arxiv_id":"2409.02076","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spinning-the-golden-thread-benchmarking-long#ran","syntology_url":"https://syntology.ai/paper/2409.02076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02076"}},"official":{"repos":["mozhu621/SGT","mozhu621/longgenbench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vhakg-a-multi-modal-knowledge-graph-based-on","slug":"vhakg-a-multi-modal-knowledge-graph-based-on","title":"VHAKG: A Multi-modal Knowledge Graph Based on Synchronized Multi-view Videos of Daily Activities","date":"2024-08-27","arxiv_id":"2408.14895","repositories_listed":2,"syntology":null},{"url":"/paper/padetbench-towards-benchmarking-physical","slug":"padetbench-towards-benchmarking-physical","title":"PADetBench: Towards Benchmarking Physical Attacks against Object Detection","date":"2024-08-17","arxiv_id":"2408.09181","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-tree-species-classification-from","slug":"benchmarking-tree-species-classification-from","title":"Benchmarking tree species classification from proximally-sensed laser scanning data: introducing the FOR-species20K dataset","date":"2024-08-12","arxiv_id":"2408.06507","repositories_listed":2,"syntology":null},{"url":"/paper/abdomenatlas-a-large-scale-detailed-annotated","slug":"abdomenatlas-a-large-scale-detailed-annotated","title":"AbdomenAtlas: A Large-Scale, Detailed-Annotated, & Multi-Center Dataset for Efficient Transfer Learning and Open Algorithmic Benchmarking","date":"2024-07-23","arxiv_id":"2407.16697","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/abdomenatlas-a-large-scale-detailed-annotated#ran","syntology_url":"https://syntology.ai/paper/2407.16697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16697"}},"official":{"repos":["mrgiovanni/abdomenatlas","mrgiovanni/suprem"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ecco-can-we-improve-model-generated-code","slug":"ecco-can-we-improve-model-generated-code","title":"ECCO: Can We Improve Model-Generated Code Efficiency Without Sacrificing Functional Correctness?","date":"2024-07-19","arxiv_id":"2407.14044","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ecco-can-we-improve-model-generated-code#ran","syntology_url":"https://syntology.ai/paper/2407.14044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14044"}},"official":{"repos":["codeeff/ecco"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/beyond-correctness-benchmarking-multi","slug":"beyond-correctness-benchmarking-multi","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","date":"2024-07-16","arxiv_id":"2407.11470","repositories_listed":2,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/beyond-correctness-benchmarking-multi#ran","syntology_url":"https://syntology.ai/paper/2407.11470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11470"}},"official":{"repos":["jszheng21/race"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/retrospective-for-the-dynamic-sensorium","slug":"retrospective-for-the-dynamic-sensorium","title":"Retrospective for the Dynamic Sensorium Competition for predicting large-scale mouse primary visual cortex activity from videos","date":"2024-07-12","arxiv_id":"2407.09100","repositories_listed":2,"syntology":{"n":21,"n_ran":17,"n_constructed":0,"n_ran_checked":16,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":11,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/retrospective-for-the-dynamic-sensorium#ran","syntology_url":"https://syntology.ai/paper/2407.09100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09100"}},"official":{"repos":["ecker-lab/sensorium_2023","bryanlimy/ViV1T"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/sh17-a-dataset-for-human-safety-and-personal","slug":"sh17-a-dataset-for-human-safety-and-personal","title":"SH17: A Dataset for Human Safety and Personal Protective Equipment Detection in Manufacturing Industry","date":"2024-07-05","arxiv_id":"2407.04590","repositories_listed":2,"syntology":null},{"url":"/paper/comics-datasets-framework-mix-of-comics","slug":"comics-datasets-framework-mix-of-comics","title":"Comics Datasets Framework: Mix of Comics datasets for detection benchmarking","date":"2024-07-03","arxiv_id":"2407.03540","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/comics-datasets-framework-mix-of-comics#ran","syntology_url":"https://syntology.ai/paper/2407.03540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03540"}},"official":{"repos":["emanuelevivoli/cdf"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/finesure-fine-grained-summarization","slug":"finesure-fine-grained-summarization","title":"FineSurE: Fine-grained Summarization Evaluation using LLMs","date":"2024-07-01","arxiv_id":"2407.00908","repositories_listed":2,"syntology":null},{"url":"/paper/ragnarok-a-reusable-rag-framework-and","slug":"ragnarok-a-reusable-rag-framework-and","title":"Ragnarök: A Reusable RAG Framework and Baselines for TREC 2024 Retrieval-Augmented Generation Track","date":"2024-06-24","arxiv_id":"2406.16828","repositories_listed":2,"syntology":{"n":23,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ragnarok-a-reusable-rag-framework-and#ran","syntology_url":"https://syntology.ai/paper/2406.16828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16828"}},"official":{"repos":["castorini/ragnarok"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/navsim-data-driven-non-reactive-autonomous","slug":"navsim-data-driven-non-reactive-autonomous","title":"NAVSIM: Data-Driven Non-Reactive Autonomous Vehicle Simulation and Benchmarking","date":"2024-06-21","arxiv_id":"2406.15349","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/navsim-data-driven-non-reactive-autonomous#ran","syntology_url":"https://syntology.ai/paper/2406.15349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15349"}},"official":{"repos":["autonomousvision/navsim"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chemfi-a-multifidelity-dataset-of-quantum","slug":"chemfi-a-multifidelity-dataset-of-quantum","title":"QeMFi: A Multifidelity Dataset of Quantum Chemical Properties of Diverse Molecules","date":"2024-06-20","arxiv_id":"2406.14149","repositories_listed":2,"syntology":null},{"url":"/paper/hotpp-benchmark-are-we-good-at-the-long","slug":"hotpp-benchmark-are-we-good-at-the-long","title":"HoTPP Benchmark: Are We Good at the Long Horizon Events Forecasting?","date":"2024-06-20","arxiv_id":"2406.14341","repositories_listed":2,"syntology":null},{"url":"/paper/genai-bench-evaluating-and-improving","slug":"genai-bench-evaluating-and-improving","title":"GenAI-Bench: Evaluating and Improving Compositional Text-to-Visual Generation","date":"2024-06-19","arxiv_id":"2406.13743","repositories_listed":2,"syntology":null},{"url":"/paper/vane-bench-video-anomaly-evaluation-benchmark","slug":"vane-bench-video-anomaly-evaluation-benchmark","title":"VANE-Bench: Video Anomaly Evaluation Benchmark for Conversational LMMs","date":"2024-06-14","arxiv_id":"2406.10326","repositories_listed":2,"syntology":null},{"url":"/paper/bag-of-tricks-benchmarking-of-jailbreak","slug":"bag-of-tricks-benchmarking-of-jailbreak","title":"Bag of Tricks: Benchmarking of Jailbreak Attacks on LLMs","date":"2024-06-13","arxiv_id":"2406.09324","repositories_listed":2,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bag-of-tricks-benchmarking-of-jailbreak#ran","syntology_url":"https://syntology.ai/paper/2406.09324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09324"}},"official":{"repos":["usail-hkust/bag_of_tricks_for_llm_jailbreaking","usail-hkust/jailtrickbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cudrt-benchmarking-the-detection-of-human-vs","slug":"cudrt-benchmarking-the-detection-of-human-vs","title":"Towards Reliable Detection of LLM-Generated Texts: A Comprehensive Evaluation Framework with CUDRT","date":"2024-06-13","arxiv_id":"2406.09056","repositories_listed":2,"syntology":null},{"url":"/paper/drivaernet-a-large-scale-multimodal-car","slug":"drivaernet-a-large-scale-multimodal-car","title":"DrivAerNet++: A Large-Scale Multimodal Car Dataset with Computational Fluid Dynamics Simulations and Deep Learning Benchmarks","date":"2024-06-13","arxiv_id":"2406.09624","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drivaernet-a-large-scale-multimodal-car#ran","syntology_url":"https://syntology.ai/paper/2406.09624","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09624"}},"official":{"repos":["mohamedelrefaie/drivaernet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-head-rag-solving-multi-aspect-problems","slug":"multi-head-rag-solving-multi-aspect-problems","title":"Multi-Head RAG: Solving Multi-Aspect Problems with LLMs","date":"2024-06-07","arxiv_id":"2406.05085","repositories_listed":2,"syntology":null},{"url":"/paper/visionad-a-software-package-of-performant","slug":"visionad-a-software-package-of-performant","title":"VisionAD, a software package of performant anomaly detection algorithms, and Proportion Localised, an interpretable metric","date":"2024-06-07","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/mars-benchmarking-the-metaphysical-reasoning","slug":"mars-benchmarking-the-metaphysical-reasoning","title":"MARS: Benchmarking the Metaphysical Reasoning Abilities of Language Models with a Multi-task Evaluation Dataset","date":"2024-06-04","arxiv_id":"2406.02106","repositories_listed":2,"syntology":null},{"url":"/paper/trutheval-a-dataset-to-evaluate-llm","slug":"trutheval-a-dataset-to-evaluate-llm","title":"TruthEval: A Dataset to Evaluate LLM Truthfulness and Reliability","date":"2024-06-04","arxiv_id":"2406.01855","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-large-language-models-on-cflue-a","slug":"benchmarking-large-language-models-on-cflue-a","title":"Benchmarking Large Language Models on CFLUE -- A Chinese Financial Language Understanding Evaluation Dataset","date":"2024-05-17","arxiv_id":"2405.10542","repositories_listed":2,"syntology":null},{"url":"/paper/polyglotoxicityprompts-multilingual","slug":"polyglotoxicityprompts-multilingual","title":"PolygloToxicityPrompts: Multilingual Evaluation of Neural Toxic Degeneration in Large Language Models","date":"2024-05-15","arxiv_id":"2405.09373","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/polyglotoxicityprompts-multilingual#ran","syntology_url":"https://syntology.ai/paper/2405.09373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.09373"}},"official":{"repos":["kpriyanshu256/polyglo-toxicity-prompts","rijgersberg/geitje"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/acegen-reinforcement-learning-of-generative","slug":"acegen-reinforcement-learning-of-generative","title":"ACEGEN: Reinforcement learning of generative chemical agents for drug discovery","date":"2024-05-07","arxiv_id":"2405.04657","repositories_listed":2,"syntology":null},{"url":"/paper/isearle-improving-textual-inversion-for-zero","slug":"isearle-improving-textual-inversion-for-zero","title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval","date":"2024-05-05","arxiv_id":"2405.02951","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/isearle-improving-textual-inversion-for-zero#ran","syntology_url":"https://syntology.ai/paper/2405.02951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02951"}},"official":{"repos":["miccunifi/circo"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/the-role-of-model-architecture-and-scale-in","slug":"the-role-of-model-architecture-and-scale-in","title":"The Role of Model Architecture and Scale in Predicting Molecular Properties: Insights from Fine-Tuning RoBERTa, BART, and LLaMA","date":"2024-05-02","arxiv_id":"2405.00949","repositories_listed":2,"syntology":null}],"record_sha256":"f06593cb38dab4c40c9fcc2cb64d500b00fdf18ea8351bd8f3a580417c2f4038","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}