{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/14","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":56,"rows_per_page":100,"rows":[1301,1400],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/13","next":"/task/benchmarking/papers/15","papers":[{"url":"/paper/icu-sepsis-a-benchmark-mdp-built-from-real","slug":"icu-sepsis-a-benchmark-mdp-built-from-real","title":"ICU-Sepsis: A Benchmark MDP Built from Real Medical Data","date":"2024-06-09","arxiv_id":"2406.05646","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/icu-sepsis-a-benchmark-mdp-built-from-real#ran","syntology_url":"https://syntology.ai/paper/2406.05646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05646"}},"official":{"repos":["icu-sepsis/icu-sepsis"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/qgeval-a-benchmark-for-question-generation","slug":"qgeval-a-benchmark-for-question-generation","title":"QGEval: Benchmarking Multi-dimensional Evaluation for Question Generation","date":"2024-06-09","arxiv_id":"2406.05707","repositories_listed":1,"syntology":null},{"url":"/paper/smiles2dock-an-open-large-scale-multi-task","slug":"smiles2dock-an-open-large-scale-multi-task","title":"Smiles2Dock: an open large-scale multi-task dataset for ML-based molecular docking","date":"2024-06-09","arxiv_id":"2406.05738","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-deep-jansen-rit-parameter","slug":"benchmarking-deep-jansen-rit-parameter","title":"Deep Jansen-Rit Parameter Inference for Model-Driven Analysis of Brain Activity","date":"2024-06-07","arxiv_id":"2406.05002","repositories_listed":1,"syntology":null},{"url":"/paper/clog-benchmarking-continual-learning-of-image","slug":"clog-benchmarking-continual-learning-of-image","title":"CLoG: Benchmarking Continual Learning of Image Generation Models","date":"2024-06-07","arxiv_id":"2406.04584","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clog-benchmarking-continual-learning-of-image#ran","syntology_url":"https://syntology.ai/paper/2406.04584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04584"}},"official":{"repos":["linhaowei1/clog"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/wildbench-benchmarking-llms-with-challenging","slug":"wildbench-benchmarking-llms-with-challenging","title":"WildBench: Benchmarking LLMs with Challenging Tasks from Real Users in the Wild","date":"2024-06-07","arxiv_id":"2406.04770","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/wildbench-benchmarking-llms-with-challenging#ran","syntology_url":"https://syntology.ai/paper/2406.04770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04770"}},"official":{"repos":["allenai/wildbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/better-late-than-never-formulating-and","slug":"better-late-than-never-formulating-and","title":"Better Late Than Never: Formulating and Benchmarking Recommendation Editing","date":"2024-06-06","arxiv_id":"2406.04553","repositories_listed":1,"syntology":null},{"url":"/paper/time-sensitive-knowledge-editing-through","slug":"time-sensitive-knowledge-editing-through","title":"Time Sensitive Knowledge Editing through Efficient Finetuning","date":"2024-06-06","arxiv_id":"2406.04496","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/time-sensitive-knowledge-editing-through#ran","syntology_url":"https://syntology.ai/paper/2406.04496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04496"}},"official":{"repos":["hiyouga/llama-factory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ader-a-comprehensive-benchmark-for-multi","slug":"ader-a-comprehensive-benchmark-for-multi","title":"A Comprehensive Library for Benchmarking Multi-class Visual Anomaly Detection","date":"2024-06-05","arxiv_id":"2406.03262","repositories_listed":1,"syntology":null},{"url":"/paper/cattleface-rgbt-rgb-t-cattle-facial-landmark","slug":"cattleface-rgbt-rgb-t-cattle-facial-landmark","title":"CattleFace-RGBT: RGB-T Cattle Facial Landmark Benchmark","date":"2024-06-05","arxiv_id":"2406.03431","repositories_listed":1,"syntology":null},{"url":"/paper/commonpower-supercharging-machine-learning","slug":"commonpower-supercharging-machine-learning","title":"CommonPower: A Framework for Safe Data-Driven Smart Grid Control","date":"2024-06-05","arxiv_id":"2406.03231","repositories_listed":1,"syntology":null},{"url":"/paper/tidmad-time-series-dataset-for-discovering","slug":"tidmad-time-series-dataset-for-discovering","title":"TIDMAD: Time Series Dataset for Discovering Dark Matter with AI Denoising","date":"2024-06-05","arxiv_id":"2406.04378","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":8,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 3 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/tidmad-time-series-dataset-for-discovering#ran","syntology_url":"https://syntology.ai/paper/2406.04378","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04378"}},"official":{"repos":["jessicafry/tidmad"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-the-feature-extractor-networks-for","slug":"analyzing-the-feature-extractor-networks-for","title":"Analyzing the Feature Extractor Networks for Face Image Synthesis","date":"2024-06-04","arxiv_id":"2406.02153","repositories_listed":1,"syntology":null},{"url":"/paper/hyperbolic-benchmarking-unveils-network","slug":"hyperbolic-benchmarking-unveils-network","title":"Hyperbolic Benchmarking Unveils Network Topology-Feature Relationship in GNN Performance","date":"2024-06-04","arxiv_id":"2406.02772","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-accord-closing-the-commonsense","slug":"texttt-accord-closing-the-commonsense","title":"$\\texttt{ACCORD}$: Closing the Commonsense Measurability Gap","date":"2024-06-04","arxiv_id":"2406.02804","repositories_listed":1,"syntology":null},{"url":"/paper/animal2vec-and-meerkat-a-self-supervised","slug":"animal2vec-and-meerkat-a-self-supervised","title":"animal2vec and MeerKAT: A self-supervised transformer for rare-event raw audio input and a large-scale reference dataset for bioacoustics","date":"2024-06-03","arxiv_id":"2406.01253","repositories_listed":1,"syntology":null},{"url":"/paper/tcmbench-a-comprehensive-benchmark-for","slug":"tcmbench-a-comprehensive-benchmark-for","title":"TCMBench: A Comprehensive Benchmark for Evaluating Large Language Models in Traditional Chinese Medicine","date":"2024-06-03","arxiv_id":"2406.01126","repositories_listed":1,"syntology":null},{"url":"/paper/genbench-a-benchmarking-suite-for-systematic","slug":"genbench-a-benchmarking-suite-for-systematic","title":"GenBench: A Benchmarking Suite for Systematic Evaluation of Genomic Foundation Models","date":"2024-06-01","arxiv_id":"2406.01627","repositories_listed":1,"syntology":null},{"url":"/paper/websuite-systematically-evaluating-why-web","slug":"websuite-systematically-evaluating-why-web","title":"WebSuite: Systematically Evaluating Why Web Agents Fail","date":"2024-06-01","arxiv_id":"2406.01623","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/websuite-systematically-evaluating-why-web#ran","syntology_url":"https://syntology.ai/paper/2406.01623","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01623"}},"official":{"repos":["erichli1/websuite"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/aquatic-navigation-a-challenging-benchmark","slug":"aquatic-navigation-a-challenging-benchmark","title":"Aquatic Navigation: A Challenging Benchmark for Deep Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.20534","repositories_listed":1,"syntology":null},{"url":"/paper/cosy-evaluating-textual-explanations-of","slug":"cosy-evaluating-textual-explanations-of","title":"CoSy: Evaluating Textual Explanations of Neurons","date":"2024-05-30","arxiv_id":"2405.20331","repositories_listed":1,"syntology":null},{"url":"/paper/llmgeo-benchmarking-large-language-models-on","slug":"llmgeo-benchmarking-large-language-models-on","title":"LLMGeo: Benchmarking Large Language Models on Image Geolocation In-the-wild","date":"2024-05-30","arxiv_id":"2405.20363","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":15,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/llmgeo-benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2405.20363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20363"}},"official":{"repos":["yeyimilk/llmgeo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/secure-benchmarking-generative-large-language","slug":"secure-benchmarking-generative-large-language","title":"SECURE: Benchmarking Large Language Models for Cybersecurity","date":"2024-05-30","arxiv_id":"2405.20441","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-improving-detail-image","slug":"benchmarking-and-improving-detail-image","title":"Benchmarking and Improving Detail Image Caption","date":"2024-05-29","arxiv_id":"2405.19092","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-improving-detail-image#ran","syntology_url":"https://syntology.ai/paper/2405.19092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19092"}},"official":{"repos":["foundation-multimodal-models/capture"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mathchat-benchmarking-mathematical-reasoning","slug":"mathchat-benchmarking-mathematical-reasoning","title":"MathChat: Benchmarking Mathematical Reasoning and Instruction Following in Multi-Turn Interactions","date":"2024-05-29","arxiv_id":"2405.19444","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathchat-benchmarking-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2405.19444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19444"}},"official":{"repos":["zhenwen-nlp/mathchat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantitative-certification-of-bias-in-large","slug":"quantitative-certification-of-bias-in-large","title":"Quantitative Certification of Bias in Large Language Models","date":"2024-05-29","arxiv_id":"2405.18780","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/quantitative-certification-of-bias-in-large#ran","syntology_url":"https://syntology.ai/paper/2405.18780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18780"}},"official":{"repos":["uiuc-focal-lab/quacer-b"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/benchmarking-skeleton-based-motion-encoder","slug":"benchmarking-skeleton-based-motion-encoder","title":"Benchmarking Skeleton-based Motion Encoder Models for Clinical Applications: Estimating Parkinson's Disease Severity in Walking Sequences","date":"2024-05-28","arxiv_id":"2405.17817","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-skeleton-based-motion-encoder#ran","syntology_url":"https://syntology.ai/paper/2405.17817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17817"}},"official":{"repos":["taatiteam/motionencoders_parkinsonism_benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dtr-bench-an-in-silico-environment-and","slug":"dtr-bench-an-in-silico-environment-and","title":"DTR-Bench: An in silico Environment and Benchmark Platform for Reinforcement Learning Based Dynamic Treatment Regime","date":"2024-05-28","arxiv_id":"2405.18610","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-improving-bird-s-eye-view","slug":"benchmarking-and-improving-bird-s-eye-view","title":"Benchmarking and Improving Bird's Eye View Perception Robustness in Autonomous Driving","date":"2024-05-27","arxiv_id":"2405.17426","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-and-improving-bird-s-eye-view#ran","syntology_url":"https://syntology.ai/paper/2405.17426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17426"}},"official":{"repos":["Daniel-xsy/RoboBEV"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lora-xs-low-rank-adaptation-with-extremely","slug":"lora-xs-low-rank-adaptation-with-extremely","title":"LoRA-XS: Low-Rank Adaptation with Extremely Small Number of Parameters","date":"2024-05-27","arxiv_id":"2405.17604","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lora-xs-low-rank-adaptation-with-extremely#ran","syntology_url":"https://syntology.ai/paper/2405.17604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17604"}},"official":{"repos":["mohammadrezabanaei/lora-xs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nuwats-mending-every-incomplete-time-series","slug":"nuwats-mending-every-incomplete-time-series","title":"NuwaTS: a Foundation Model Mending Every Incomplete Time Series","date":"2024-05-24","arxiv_id":"2405.15317","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuwats-mending-every-incomplete-time-series#ran","syntology_url":"https://syntology.ai/paper/2405.15317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15317"}},"official":{"repos":["chengyui/nuwats"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-training-state-of-the","slug":"an-empirical-study-of-training-state-of-the","title":"An Empirical Study of Training State-of-the-Art LiDAR Segmentation Models","date":"2024-05-23","arxiv_id":"2405.14870","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-training-state-of-the#ran","syntology_url":"https://syntology.ai/paper/2405.14870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14870"}},"official":{"repos":["open-mmlab/mmdetection3d"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analog-or-digital-in-memory-computing","slug":"analog-or-digital-in-memory-computing","title":"Analog or Digital In-memory Computing? Benchmarking through Quantitative Modeling","date":"2024-05-23","arxiv_id":"2405.14978","repositories_listed":1,"syntology":null},{"url":"/paper/androidworld-a-dynamic-benchmarking","slug":"androidworld-a-dynamic-benchmarking","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","date":"2024-05-23","arxiv_id":"2405.14573","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/androidworld-a-dynamic-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2405.14573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14573"}},"official":{"repos":["google-research/android_world"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gcondenser-benchmarking-graph-condensation","slug":"gcondenser-benchmarking-graph-condensation","title":"GCondenser: Benchmarking Graph Condensation","date":"2024-05-23","arxiv_id":"2405.14246","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gcondenser-benchmarking-graph-condensation#ran","syntology_url":"https://syntology.ai/paper/2405.14246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14246"}},"official":{"repos":["superallen13/GCondenser"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/s-eval-automatic-and-adaptive-test-generation","slug":"s-eval-automatic-and-adaptive-test-generation","title":"S-Eval: Towards Automated and Comprehensive Safety Evaluation for Large Language Models","date":"2024-05-23","arxiv_id":"2405.14191","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-fish-dataset-and-evaluation","slug":"benchmarking-fish-dataset-and-evaluation","title":"Benchmarking Fish Dataset and Evaluation Metric in Keypoint Detection -- Towards Precise Fish Morphological Assessment in Aquaculture Breeding","date":"2024-05-21","arxiv_id":"2405.12476","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/benchmarking-fish-dataset-and-evaluation#ran","syntology_url":"https://syntology.ai/paper/2405.12476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12476"}},"official":{"repos":["weizhenliubioinform/fish-phenotype-detect"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/disparisk-assessing-and-interpreting","slug":"disparisk-assessing-and-interpreting","title":"DispaRisk: Auditing Fairness Through Usable Information","date":"2024-05-20","arxiv_id":"2405.12372","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-multi-center-ct-and-mri","slug":"large-scale-multi-center-ct-and-mri","title":"Large-Scale Multi-Center CT and MRI Segmentation of Pancreas with Deep Learning","date":"2024-05-20","arxiv_id":"2405.12367","repositories_listed":1,"syntology":null},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-integrated-framework-for-multi-granular","slug":"an-integrated-framework-for-multi-granular","title":"An Integrated Framework for Multi-Granular Explanation of Video Summarization","date":"2024-05-16","arxiv_id":"2405.10082","repositories_listed":1,"syntology":null},{"url":"/paper/documint-docstring-generation-for-python","slug":"documint-docstring-generation-for-python","title":"DocuMint: Docstring Generation for Python using Small Language Models","date":"2024-05-16","arxiv_id":"2405.10243","repositories_listed":1,"syntology":null},{"url":"/paper/simulation-based-benchmarking-of","slug":"simulation-based-benchmarking-of","title":"Simulation-Based Benchmarking of Reinforcement Learning Agents for Personalized Retail Promotions","date":"2024-05-16","arxiv_id":"2405.10469","repositories_listed":1,"syntology":null},{"url":"/paper/scifibench-benchmarking-large-multimodal","slug":"scifibench-benchmarking-large-multimodal","title":"SciFIBench: Benchmarking Large Multimodal Models for Scientific Figure Interpretation","date":"2024-05-14","arxiv_id":"2405.08807","repositories_listed":1,"syntology":null},{"url":"/paper/divergent-creativity-in-humans-and-large","slug":"divergent-creativity-in-humans-and-large","title":"Divergent Creativity in Humans and Large Language Models","date":"2024-05-13","arxiv_id":"2405.13012","repositories_listed":1,"syntology":null},{"url":"/paper/noisebench-benchmarking-the-impact-of-real","slug":"noisebench-benchmarking-the-impact-of-real","title":"NoiseBench: Benchmarking the Impact of Real Label Noise on Named Entity Recognition","date":"2024-05-13","arxiv_id":"2405.07609","repositories_listed":1,"syntology":null},{"url":"/paper/replication-study-and-benchmarking-of-real","slug":"replication-study-and-benchmarking-of-real","title":"Replication Study and Benchmarking of Real-Time Object Detection Models","date":"2024-05-11","arxiv_id":"2405.06911","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-classical-and-learning-based","slug":"benchmarking-classical-and-learning-based","title":"Benchmarking Classical and Learning-Based Multibeam Point Cloud Registration","date":"2024-05-10","arxiv_id":"2405.06279","repositories_listed":1,"syntology":null},{"url":"/paper/aequitas-flow-streamlining-fair-ml","slug":"aequitas-flow-streamlining-fair-ml","title":"Aequitas Flow: Streamlining Fair ML Experimentation","date":"2024-05-09","arxiv_id":"2405.05809","repositories_listed":1,"syntology":null},{"url":"/paper/llm-qbench-a-benchmark-towards-the-best","slug":"llm-qbench-a-benchmark-towards-the-best","title":"LLMC: Benchmarking Large Language Model Quantization with a Versatile Compression Toolkit","date":"2024-05-09","arxiv_id":"2405.06001","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/llm-qbench-a-benchmark-towards-the-best#ran","syntology_url":"https://syntology.ai/paper/2405.06001","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06001"}},"official":{"repos":["modeltc/llmc"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/benchmarking-educational-program-repair","slug":"benchmarking-educational-program-repair","title":"Benchmarking Educational Program Repair","date":"2024-05-08","arxiv_id":"2405.05347","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-educational-program-repair#ran","syntology_url":"https://syntology.ai/paper/2405.05347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05347"}},"official":{"repos":["koutchemecharles/gaied_nips23"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-in-lung-health-benchmarking-detection-and","slug":"ai-in-lung-health-benchmarking-detection-and","title":"AI in Lung Health: Benchmarking Detection and Diagnostic Models Across Multiple CT Scan Datasets","date":"2024-05-07","arxiv_id":"2405.04605","repositories_listed":1,"syntology":null},{"url":"/paper/refining-joint-text-and-source-code","slug":"refining-joint-text-and-source-code","title":"Refining Joint Text and Source Code Embeddings for Retrieval Task with Parameter-Efficient Fine-Tuning","date":"2024-05-07","arxiv_id":"2405.04126","repositories_listed":1,"syntology":null},{"url":"/paper/performance-evaluation-of-real-time-object","slug":"performance-evaluation-of-real-time-object","title":"Performance Evaluation of Real-Time Object Detection for Electric Scooters","date":"2024-05-05","arxiv_id":"2405.03039","repositories_listed":1,"syntology":null},{"url":"/paper/position-paper-quo-vadis-unsupervised-time","slug":"position-paper-quo-vadis-unsupervised-time","title":"Position: Quo Vadis, Unsupervised Time Series Anomaly Detection?","date":"2024-05-04","arxiv_id":"2405.02678","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-paper-quo-vadis-unsupervised-time#ran","syntology_url":"https://syntology.ai/paper/2405.02678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.02678"}},"official":{"repos":["ssarfraz/QuoVadisTAD"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/single-and-multi-hop-question-answering","slug":"single-and-multi-hop-question-answering","title":"Single and Multi-Hop Question-Answering Datasets for Reticular Chemistry with GPT-4-Turbo","date":"2024-05-03","arxiv_id":"2405.02128","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-representations-for-speech-music","slug":"benchmarking-representations-for-speech-music","title":"Benchmarking Representations for Speech, Music, and Acoustic Events","date":"2024-05-02","arxiv_id":"2405.00934","repositories_listed":1,"syntology":null},{"url":"/paper/hlsfactory-a-framework-empowering-high-level","slug":"hlsfactory-a-framework-empowering-high-level","title":"HLSFactory: A Framework Empowering High-Level Synthesis Datasets for Machine Learning and Beyond","date":"2024-05-01","arxiv_id":"2405.00820","repositories_listed":1,"syntology":null},{"url":"/paper/atommic-an-advanced-toolbox-for-multitask","slug":"atommic-an-advanced-toolbox-for-multitask","title":"ATOMMIC: An Advanced Toolbox for Multitask Medical Imaging Consistency to facilitate Artificial Intelligence applications from acquisition to analysis in Magnetic Resonance Imaging","date":"2024-04-30","arxiv_id":"2404.19665","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-benchmark-leakage-in-large","slug":"benchmarking-benchmark-leakage-in-large","title":"Benchmarking Benchmark Leakage in Large Language Models","date":"2024-04-29","arxiv_id":"2404.18824","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-benchmark-leakage-in-large#ran","syntology_url":"https://syntology.ai/paper/2404.18824","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18824"}},"official":{"repos":["gair-nlp/benbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-critical-treatment-effect-bias-in","slug":"detecting-critical-treatment-effect-bias-in","title":"Detecting critical treatment effect bias in small subgroups","date":"2024-04-29","arxiv_id":"2404.18905","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/detecting-critical-treatment-effect-bias-in#ran","syntology_url":"https://syntology.ai/paper/2404.18905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18905"}},"official":{"repos":["jaabmar/kernel-test-bias"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-vision-language-decoders-use-images-and","slug":"do-vision-language-decoders-use-images-and","title":"Do Vision & Language Decoders use Images and Text equally? How Self-consistent are their Explanations?","date":"2024-04-29","arxiv_id":"2404.18624","repositories_listed":1,"syntology":null},{"url":"/paper/leak-proof-cmap-a-framework-for-training-and","slug":"leak-proof-cmap-a-framework-for-training-and","title":"Leak Proof CMap; a framework for training and evaluation of cell line agnostic L1000 similarity methods","date":"2024-04-29","arxiv_id":"2404.18960","repositories_listed":1,"syntology":null},{"url":"/paper/sidbench-a-python-framework-for-reliably","slug":"sidbench-a-python-framework-for-reliably","title":"SIDBench: A Python Framework for Reliably Assessing Synthetic Image Detection Methods","date":"2024-04-29","arxiv_id":"2404.18552","repositories_listed":1,"syntology":null},{"url":"/paper/4dbinfer-a-4d-benchmarking-toolbox-for-graph","slug":"4dbinfer-a-4d-benchmarking-toolbox-for-graph","title":"4DBInfer: A 4D Benchmarking Toolbox for Graph-Centric Predictive Modeling on Relational DBs","date":"2024-04-28","arxiv_id":"2404.18209","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/4dbinfer-a-4d-benchmarking-toolbox-for-graph#ran","syntology_url":"https://syntology.ai/paper/2404.18209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18209"}},"official":{"repos":["awslabs/multi-table-benchmark"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/constellation-dataset-benchmarking-high","slug":"constellation-dataset-benchmarking-high","title":"Constellation Dataset: Benchmarking High-Altitude Object Detection for an Urban Intersection","date":"2024-04-25","arxiv_id":"2404.16944","repositories_listed":1,"syntology":null},{"url":"/paper/crisp-leveraging-tread-depth-maps-for","slug":"crisp-leveraging-tread-depth-maps-for","title":"CriSp: Leveraging Tread Depth Maps for Enhanced Crime-Scene Shoeprint Matching","date":"2024-04-25","arxiv_id":"2404.16972","repositories_listed":1,"syntology":null},{"url":"/paper/apistox-a-new-benchmark-dataset-for-the","slug":"apistox-a-new-benchmark-dataset-for-the","title":"ApisTox: a new benchmark dataset for the classification of small molecules toxicity on honey bees","date":"2024-04-24","arxiv_id":"2404.16196","repositories_listed":1,"syntology":null},{"url":"/paper/implicitave-an-open-source-dataset-and","slug":"implicitave-an-open-source-dataset-and","title":"ImplicitAVE: An Open-Source Dataset and Multimodal LLMs Benchmark for Implicit Attribute Value Extraction","date":"2024-04-24","arxiv_id":"2404.15592","repositories_listed":1,"syntology":null},{"url":"/paper/syntheval-a-framework-for-detailed-utility","slug":"syntheval-a-framework-for-detailed-utility","title":"SynthEval: A Framework for Detailed Utility and Privacy Evaluation of Tabular Synthetic Data","date":"2024-04-24","arxiv_id":"2404.15821","repositories_listed":1,"syntology":null},{"url":"/paper/importance-of-disjoint-sampling-in","slug":"importance-of-disjoint-sampling-in","title":"Importance of Disjoint Sampling in Conventional and Transformer Models for Hyperspectral Image Classification","date":"2024-04-23","arxiv_id":"2404.14944","repositories_listed":1,"syntology":null},{"url":"/paper/a-user-centric-benchmark-for-evaluating-large","slug":"a-user-centric-benchmark-for-evaluating-large","title":"A User-Centric Multi-Intent Benchmark for Evaluating Large Language Models","date":"2024-04-22","arxiv_id":"2404.13940","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-user-centric-benchmark-for-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2404.13940","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13940"}},"official":{"repos":["alice1998/urs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/experimental-validation-of-ultrasound","slug":"experimental-validation-of-ultrasound","title":"Experimental Validation of Ultrasound Beamforming with End-to-End Deep Learning for Single Plane Wave Imaging","date":"2024-04-22","arxiv_id":"2404.14188","repositories_listed":1,"syntology":null},{"url":"/paper/tavgbench-benchmarking-text-to-audible-video","slug":"tavgbench-benchmarking-text-to-audible-video","title":"TAVGBench: Benchmarking Text to Audible-Video Generation","date":"2024-04-22","arxiv_id":"2404.14381","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":5,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tavgbench-benchmarking-text-to-audible-video#ran","syntology_url":"https://syntology.ai/paper/2404.14381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14381"}},"official":{"repos":["opennlplab/tavgbench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepfake-o-meter-v2-0-an-open-platform-for","slug":"deepfake-o-meter-v2-0-an-open-platform-for","title":"DeepFake-O-Meter v2.0: An Open Platform for DeepFake Detection","date":"2024-04-19","arxiv_id":"2404.13146","repositories_listed":1,"syntology":null},{"url":"/paper/rexel-an-end-to-end-model-for-document-level","slug":"rexel-an-end-to-end-model-for-document-level","title":"REXEL: An End-to-end Model for Document-Level Relation Extraction and Entity Linking","date":"2024-04-19","arxiv_id":"2404.12788","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rexel-an-end-to-end-model-for-document-level#ran","syntology_url":"https://syntology.ai/paper/2404.12788","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12788"}},"official":{"repos":["amazon-science/e2e-docie"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/stark-benchmarking-llm-retrieval-on-textual","slug":"stark-benchmarking-llm-retrieval-on-textual","title":"STaRK: Benchmarking LLM Retrieval on Textual and Relational Knowledge Bases","date":"2024-04-19","arxiv_id":"2404.13207","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stark-benchmarking-llm-retrieval-on-textual#ran","syntology_url":"https://syntology.ai/paper/2404.13207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13207"}},"official":{"repos":["snap-stanford/stark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/longembed-extending-embedding-models-for-long","slug":"longembed-extending-embedding-models-for-long","title":"LongEmbed: Extending Embedding Models for Long Context Retrieval","date":"2024-04-18","arxiv_id":"2404.12096","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/longembed-extending-embedding-models-for-long#ran","syntology_url":"https://syntology.ai/paper/2404.12096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12096"}},"official":{"repos":["dwzhu-pku/longembed"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vbr-a-vision-benchmark-in-rome","slug":"vbr-a-vision-benchmark-in-rome","title":"VBR: A Vision Benchmark in Rome","date":"2024-04-17","arxiv_id":"2404.11322","repositories_listed":1,"syntology":null},{"url":"/paper/revealing-data-leakage-in-protein-interaction","slug":"revealing-data-leakage-in-protein-interaction","title":"Revealing data leakage in protein interaction benchmarks","date":"2024-04-16","arxiv_id":"2404.10457","repositories_listed":1,"syntology":null},{"url":"/paper/a-large-scale-evaluation-of-speech-foundation","slug":"a-large-scale-evaluation-of-speech-foundation","title":"A Large-Scale Evaluation of Speech Foundation Models","date":"2024-04-15","arxiv_id":"2404.09385","repositories_listed":1,"syntology":null},{"url":"/paper/a-recipe-for-cac-mosaic-based-generalized","slug":"a-recipe-for-cac-mosaic-based-generalized","title":"A Recipe for CAC: Mosaic-based Generalized Loss for Improved Class-Agnostic Counting","date":"2024-04-15","arxiv_id":"2404.09826","repositories_listed":1,"syntology":null},{"url":"/paper/a-review-and-efficient-implementation-of","slug":"a-review-and-efficient-implementation-of","title":"A Review and Efficient Implementation of Scene Graph Generation Metrics","date":"2024-04-15","arxiv_id":"2404.09616","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-review-and-efficient-implementation-of#ran","syntology_url":"https://syntology.ai/paper/2404.09616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09616"}},"official":{"repos":["lorjul/sgbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ampcliff-quantitative-definition-and","slug":"ampcliff-quantitative-definition-and","title":"AMPCliff: quantitative definition and benchmarking of activity cliffs in antimicrobial peptides","date":"2024-04-15","arxiv_id":"2404.09738","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llama2-mistral-gemma-and-gpt-for","slug":"benchmarking-llama2-mistral-gemma-and-gpt-for","title":"Benchmarking Llama2, Mistral, Gemma and GPT for Factuality, Toxicity, Bias and Propensity for Hallucinations","date":"2024-04-15","arxiv_id":"2404.09785","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-llama2-mistral-gemma-and-gpt-for#ran","syntology_url":"https://syntology.ai/paper/2404.09785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09785"}},"official":{"repos":["innodatalabs/innodata-llm-safety"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/from-bytes-to-borsch-fine-tuning-gemma-and","slug":"from-bytes-to-borsch-fine-tuning-gemma-and","title":"From Bytes to Borsch: Fine-Tuning Gemma and Mistral for the Ukrainian Language Representation","date":"2024-04-14","arxiv_id":"2404.09138","repositories_listed":1,"syntology":null},{"url":"/paper/roofdiffusion-constructing-roofs-from","slug":"roofdiffusion-constructing-roofs-from","title":"RoofDiffusion: Constructing Roofs from Severely Corrupted Point Data via Diffusion","date":"2024-04-14","arxiv_id":"2404.09290","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/roofdiffusion-constructing-roofs-from#ran","syntology_url":"https://syntology.ai/paper/2404.09290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09290"}},"official":{"repos":["kylelo/roofdiffusion"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/towards-sim-to-real-industrial-parts","slug":"towards-sim-to-real-industrial-parts","title":"Towards Sim-to-Real Industrial Parts Classification with Synthetic Dataset","date":"2024-04-12","arxiv_id":"2404.08778","repositories_listed":1,"syntology":null},{"url":"/paper/osworld-benchmarking-multimodal-agents-for","slug":"osworld-benchmarking-multimodal-agents-for","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","date":"2024-04-11","arxiv_id":"2404.07972","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/osworld-benchmarking-multimodal-agents-for#ran","syntology_url":"https://syntology.ai/paper/2404.07972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07972"}},"official":{"repos":["xlang-ai/OSWorld"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-your-llm-outdated-benchmarking-llms","slug":"is-your-llm-outdated-benchmarking-llms","title":"DyKnow: Dynamically Verifying Time-Sensitive Factual Knowledge in LLMs","date":"2024-04-10","arxiv_id":"2404.08700","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-your-llm-outdated-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2404.08700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08700"}},"official":{"repos":["sislab-unitn/dyknow"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accel-nasbench-sustainable-benchmarking-for","slug":"accel-nasbench-sustainable-benchmarking-for","title":"Accel-NASBench: Sustainable Benchmarking for Accelerator-Aware NAS","date":"2024-04-09","arxiv_id":"2404.08005","repositories_listed":1,"syntology":null},{"url":"/paper/agentquest-a-modular-benchmark-framework-to","slug":"agentquest-a-modular-benchmark-framework-to","title":"AgentQuest: A Modular Benchmark Framework to Measure Progress and Improve LLM Agents","date":"2024-04-09","arxiv_id":"2404.06411","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agentquest-a-modular-benchmark-framework-to#ran","syntology_url":"https://syntology.ai/paper/2404.06411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06411"}},"official":{"repos":["nec-research/agentquest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efsa-towards-event-level-financial-sentiment","slug":"efsa-towards-event-level-financial-sentiment","title":"EFSA: Towards Event-Level Financial Sentiment Analysis","date":"2024-04-08","arxiv_id":"2404.08681","repositories_listed":1,"syntology":null},{"url":"/paper/hoeg-a-new-approach-for-object-centric","slug":"hoeg-a-new-approach-for-object-centric","title":"HOEG: A New Approach for Object-Centric Predictive Process Monitoring","date":"2024-04-08","arxiv_id":"2404.05316","repositories_listed":1,"syntology":null},{"url":"/paper/towards-objectively-benchmarking-social","slug":"towards-objectively-benchmarking-social","title":"Towards Objectively Benchmarking Social Intelligence for Language Agents at Action Level","date":"2024-04-08","arxiv_id":"2404.05337","repositories_listed":1,"syntology":null},{"url":"/paper/mlake-multilingual-knowledge-editing","slug":"mlake-multilingual-knowledge-editing","title":"MLaKE: Multilingual Knowledge Editing Benchmark for Large Language Models","date":"2024-04-07","arxiv_id":"2404.04990","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-video-summarization-with-context","slug":"enhancing-video-summarization-with-context","title":"Enhancing Video Summarization with Context Awareness","date":"2024-04-06","arxiv_id":"2404.04564","repositories_listed":1,"syntology":null},{"url":"/paper/pollmgraph-unraveling-hallucinations-in-large","slug":"pollmgraph-unraveling-hallucinations-in-large","title":"PoLLMgraph: Unraveling Hallucinations in Large Language Models via State Transition Dynamics","date":"2024-04-06","arxiv_id":"2404.04722","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pollmgraph-unraveling-hallucinations-in-large#ran","syntology_url":"https://syntology.ai/paper/2404.04722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04722"}},"official":{"repos":["hitum-dev/pollmgraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-improving-compositional","slug":"benchmarking-and-improving-compositional","title":"Benchmarking and Improving Compositional Generalization of Multi-aspect Controllable Text Generation","date":"2024-04-05","arxiv_id":"2404.04232","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-improving-compositional#ran","syntology_url":"https://syntology.ai/paper/2404.04232","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04232"}},"official":{"repos":["tqzhong/cg4mctg"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/who-evaluates-the-evaluations-objectively","slug":"who-evaluates-the-evaluations-objectively","title":"Who Evaluates the Evaluations? Objectively Scoring Text-to-Image Prompt Coherence Metrics with T2IScoreScore (TS2)","date":"2024-04-05","arxiv_id":"2404.04251","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/who-evaluates-the-evaluations-objectively#ran","syntology_url":"https://syntology.ai/paper/2404.04251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04251"}},"official":{"repos":["michaelsaxon/T2IScoreScore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"f0d87cfe735228a56803779e63f8b69439cb638aa7a6755f2748d5f8e0302ca3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}