{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/9","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":56,"rows_per_page":100,"rows":[801,900],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/8","next":"/task/benchmarking/papers/10","papers":[{"url":"/paper/foundation-model-of-electronic-medical","slug":"foundation-model-of-electronic-medical","title":"Foundation Model of Electronic Medical Records for Adaptive Risk Estimation","date":"2025-02-10","arxiv_id":"2502.06124","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-for-synthetic-speech-detection","slug":"less-is-more-for-synthetic-speech-detection","title":"ShiftySpeech: A Large-Scale Synthetic Speech Dataset with Distribution Shifts","date":"2025-02-08","arxiv_id":"2502.05674","repositories_listed":1,"syntology":null},{"url":"/paper/mol-moe-training-preference-guided-routers","slug":"mol-moe-training-preference-guided-routers","title":"Mol-MoE: Training Preference-Guided Routers for Molecule Generation","date":"2025-02-08","arxiv_id":"2502.05633","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mol-moe-training-preference-guided-routers#ran","syntology_url":"https://syntology.ai/paper/2502.05633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05633"}},"official":{"repos":["ddidacus/mol-moe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-extended-benchmarking-of-multi-agent","slug":"an-extended-benchmarking-of-multi-agent","title":"An Extended Benchmarking of Multi-Agent Reinforcement Learning Algorithms in Complex Fully Cooperative Tasks","date":"2025-02-07","arxiv_id":"2502.04773","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-extended-benchmarking-of-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2502.04773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04773"}},"official":{"repos":["ailabdsunipi/pymarlzooplus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-audiobox-aesthetics-unified-automatic","slug":"meta-audiobox-aesthetics-unified-automatic","title":"Meta Audiobox Aesthetics: Unified Automatic Quality Assessment for Speech, Music, and Sound","date":"2025-02-07","arxiv_id":"2502.05139","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-multi-robot-systems","slug":"large-language-models-for-multi-robot-systems","title":"Large Language Models for Multi-Robot Systems: A Survey","date":"2025-02-06","arxiv_id":"2502.03814","repositories_listed":1,"syntology":null},{"url":"/paper/pint-physics-informed-neural-time-series","slug":"pint-physics-informed-neural-time-series","title":"PINT: Physics-Informed Neural Time Series Models with Applications to Long-term Inference on WeatherBench 2m-Temperature Data","date":"2025-02-06","arxiv_id":"2502.04018","repositories_listed":1,"syntology":null},{"url":"/paper/sok-benchmarking-poisoning-attacks-and","slug":"sok-benchmarking-poisoning-attacks-and","title":"SoK: Benchmarking Poisoning Attacks and Defenses in Federated Learning","date":"2025-02-06","arxiv_id":"2502.03801","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-datasets-for-machine-learning-on","slug":"synthetic-datasets-for-machine-learning-on","title":"Synthetic Datasets for Machine Learning on Spatio-Temporal Graphs using PDEs","date":"2025-02-06","arxiv_id":"2502.04140","repositories_listed":1,"syntology":null},{"url":"/paper/picbench-benchmarking-llms-for-photonic","slug":"picbench-benchmarking-llms-for-photonic","title":"PICBench: Benchmarking LLMs for Photonic Integrated Circuits Design","date":"2025-02-05","arxiv_id":"2502.03159","repositories_listed":1,"syntology":null},{"url":"/paper/speculative-prefill-turbocharging-ttft-with","slug":"speculative-prefill-turbocharging-ttft-with","title":"Speculative Prefill: Turbocharging TTFT with Lightweight and Training-Free Token Importance Estimation","date":"2025-02-05","arxiv_id":"2502.02789","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speculative-prefill-turbocharging-ttft-with#ran","syntology_url":"https://syntology.ai/paper/2502.02789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02789"}},"official":{"repos":["Jingyu6/speculative_prefill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tgb-seq-benchmark-challenging-temporal-gnns","slug":"tgb-seq-benchmark-challenging-temporal-gnns","title":"TGB-Seq Benchmark: Challenging Temporal GNNs with Complex Sequential Dynamics","date":"2025-02-05","arxiv_id":"2502.02975","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tgb-seq-benchmark-challenging-temporal-gnns#ran","syntology_url":"https://syntology.ai/paper/2502.02975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02975"}},"official":{"repos":["TGB-Seq/TGB-Seq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comparison-of-translation-performance","slug":"a-comparison-of-translation-performance","title":"A comparison of translation performance between DeepL and Supertext","date":"2025-02-04","arxiv_id":"2502.02577","repositories_listed":1,"syntology":null},{"url":"/paper/no-metric-to-rule-them-all-toward-principled","slug":"no-metric-to-rule-them-all-toward-principled","title":"No Metric to Rule Them All: Toward Principled Evaluations of Graph-Learning Datasets","date":"2025-02-04","arxiv_id":"2502.02379","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/no-metric-to-rule-them-all-toward-principled#ran","syntology_url":"https://syntology.ai/paper/2502.02379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02379"}},"official":{"repos":["aidos-lab/rings"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/rankify-a-comprehensive-python-toolkit-for","slug":"rankify-a-comprehensive-python-toolkit-for","title":"Rankify: A Comprehensive Python Toolkit for Retrieval, Re-Ranking, and Retrieval-Augmented Generation","date":"2025-02-04","arxiv_id":"2502.02464","repositories_listed":1,"syntology":null},{"url":"/paper/learned-bayesian-cramer-rao-bound-for-unknown","slug":"learned-bayesian-cramer-rao-bound-for-unknown","title":"Learned Bayesian Cramér-Rao Bound for Unknown Measurement Models Using Score Neural Networks","date":"2025-02-02","arxiv_id":"2502.00724","repositories_listed":1,"syntology":null},{"url":"/paper/mm-iq-benchmarking-human-like-abstraction-and-1","slug":"mm-iq-benchmarking-human-like-abstraction-and-1","title":"MM-IQ: Benchmarking Human-Like Abstraction and Reasoning in Multimodal Models","date":"2025-02-02","arxiv_id":"2502.00698","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mm-iq-benchmarking-human-like-abstraction-and-1#ran","syntology_url":"https://syntology.ai/paper/2502.00698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.00698"}},"official":{"repos":["AceCHQ/MMIQ"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-itobos-dataset-skin-region-images","slug":"the-itobos-dataset-skin-region-images","title":"The iToBoS dataset: skin region images extracted from 3D total body photographs for lesion detection","date":"2025-01-30","arxiv_id":"2501.18270","repositories_listed":1,"syntology":null},{"url":"/paper/unraveling-the-capabilities-of-language","slug":"unraveling-the-capabilities-of-language","title":"Unraveling the Capabilities of Language Models in News Summarization","date":"2025-01-30","arxiv_id":"2501.18128","repositories_listed":1,"syntology":null},{"url":"/paper/hatebench-benchmarking-hate-speech-detectors","slug":"hatebench-benchmarking-hate-speech-detectors","title":"HateBench: Benchmarking Hate Speech Detectors on LLM-Generated Content and Hate Campaigns","date":"2025-01-28","arxiv_id":"2501.16750","repositories_listed":1,"syntology":null},{"url":"/paper/saferag-benchmarking-security-in-retrieval","slug":"saferag-benchmarking-security-in-retrieval","title":"SafeRAG: Benchmarking Security in Retrieval-Augmented Generation of Large Language Model","date":"2025-01-28","arxiv_id":"2501.18636","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-quantum-reinforcement-learning","slug":"benchmarking-quantum-reinforcement-learning","title":"Benchmarking Quantum Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.15893","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2501.15893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15893"}},"official":{"repos":["nicomeyer96/qrl-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gianthunter-accurate-detection-of-giant-virus","slug":"gianthunter-accurate-detection-of-giant-virus","title":"GiantHunter: Accurate detection of giant virus in metagenomic data using reinforcement-learning and Monte Carlo tree search","date":"2025-01-26","arxiv_id":"2501.15472","repositories_listed":1,"syntology":null},{"url":"/paper/medagentbench-dataset-for-benchmarking-llms","slug":"medagentbench-dataset-for-benchmarking-llms","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","date":"2025-01-24","arxiv_id":"2501.14654","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medagentbench-dataset-for-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2501.14654","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14654"}},"official":{"repos":["stanfordmlgroup/medagentbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-benchmarking-and-robust-learning-for","slug":"scalable-benchmarking-and-robust-learning-for","title":"Scalable Benchmarking and Robust Learning for Noise-Free Ego-Motion and 3D Reconstruction from Noisy Video","date":"2025-01-24","arxiv_id":"2501.14319","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-benchmarking-and-robust-learning-for#ran","syntology_url":"https://syntology.ai/paper/2501.14319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14319"}},"official":{"repos":["xiaohao-xu/slam-under-perturbation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-biomedical-relation-extraction-with-1","slug":"enhancing-biomedical-relation-extraction-with-1","title":"Enhancing Biomedical Relation Extraction with Directionality","date":"2025-01-23","arxiv_id":"2501.14079","repositories_listed":1,"syntology":null},{"url":"/paper/does-table-source-matter-benchmarking-and","slug":"does-table-source-matter-benchmarking-and","title":"Does Table Source Matter? Benchmarking and Improving Multimodal Scientific Table Understanding and Reasoning","date":"2025-01-22","arxiv_id":"2501.13042","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-image-perturbations-for-testing","slug":"benchmarking-image-perturbations-for-testing","title":"Benchmarking Image Perturbations for Testing Automated Driving Assistance Systems","date":"2025-01-21","arxiv_id":"2501.12269","repositories_listed":1,"syntology":null},{"url":"/paper/insqabench-benchmarking-chinese-insurance","slug":"insqabench-benchmarking-chinese-insurance","title":"InsQABench: Benchmarking Chinese Insurance Domain Question Answering with Large Language Models","date":"2025-01-19","arxiv_id":"2501.10943","repositories_listed":1,"syntology":null},{"url":"/paper/colorgrid-a-multi-agent-non-stationary","slug":"colorgrid-a-multi-agent-non-stationary","title":"ColorGrid: A Multi-Agent Non-Stationary Environment for Goal Inference and Assistance","date":"2025-01-17","arxiv_id":"2501.10593","repositories_listed":1,"syntology":null},{"url":"/paper/pixelbrax-learning-continuous-control-from","slug":"pixelbrax-learning-continuous-control-from","title":"PixelBrax: Learning Continuous Control from Pixels End-to-End on the GPU","date":"2025-01-16","arxiv_id":"2502.00021","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-sat-and-smt-solvers-on-large-scale","slug":"evaluating-sat-and-smt-solvers-on-large-scale","title":"Evaluating SAT and SMT Solvers on Large-Scale Sudoku Puzzles","date":"2025-01-15","arxiv_id":"2501.08569","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-llms-can-reason-about-aesthetics","slug":"multimodal-llms-can-reason-about-aesthetics","title":"Multimodal LLMs Can Reason about Aesthetics in Zero-Shot","date":"2025-01-15","arxiv_id":"2501.09012","repositories_listed":1,"syntology":null},{"url":"/paper/tomato-verbalizing-the-mental-states-of-role","slug":"tomato-verbalizing-the-mental-states-of-role","title":"ToMATO: Verbalizing the Mental States of Role-Playing LLMs for Benchmarking Theory of Mind","date":"2025-01-15","arxiv_id":"2501.08838","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-graph-representations-and-graph","slug":"benchmarking-graph-representations-and-graph","title":"Benchmarking Graph Representations and Graph Neural Networks for Multivariate Time Series Classification","date":"2025-01-14","arxiv_id":"2501.08305","repositories_listed":1,"syntology":null},{"url":"/paper/vchitect-2-0-parallel-transformer-for-scaling","slug":"vchitect-2-0-parallel-transformer-for-scaling","title":"Vchitect-2.0: Parallel Transformer for Scaling Up Video Diffusion Models","date":"2025-01-14","arxiv_id":"2501.08453","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vchitect-2-0-parallel-transformer-for-scaling#ran","syntology_url":"https://syntology.ai/paper/2501.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08453"}},"official":null}},{"url":"/paper/stronger-than-you-think-benchmarking-weak","slug":"stronger-than-you-think-benchmarking-weak","title":"Stronger Than You Think: Benchmarking Weak Supervision on Realistic Tasks","date":"2025-01-13","arxiv_id":"2501.07727","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stronger-than-you-think-benchmarking-weak#ran","syntology_url":"https://syntology.ai/paper/2501.07727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07727"}},"official":{"repos":["jeffreywpli/stronger-than-you-think"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timbervision-a-multi-task-dataset-and","slug":"timbervision-a-multi-task-dataset-and","title":"TimberVision: A Multi-Task Dataset and Framework for Log-Component Segmentation and Tracking in Autonomous Forestry Operations","date":"2025-01-13","arxiv_id":"2501.07360","repositories_listed":1,"syntology":null},{"url":"/paper/zno-eval-benchmarking-reasoning-capabilities","slug":"zno-eval-benchmarking-reasoning-capabilities","title":"ZNO-Eval: Benchmarking reasoning capabilities of large language models in Ukrainian","date":"2025-01-12","arxiv_id":"2501.06715","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-dialogue-knowledge","slug":"retrieval-augmented-dialogue-knowledge","title":"Retrieval-Augmented Dialogue Knowledge Aggregation for Expressive Conversational Speech Synthesis","date":"2025-01-11","arxiv_id":"2501.06467","repositories_listed":1,"syntology":null},{"url":"/paper/diffusets-12-lead-ecg-generation-conditioned","slug":"diffusets-12-lead-ecg-generation-conditioned","title":"DiffuSETS: 12-lead ECG Generation Conditioned on Clinical Text Reports and Patient-Specific Information","date":"2025-01-10","arxiv_id":"2501.05932","repositories_listed":1,"syntology":null},{"url":"/paper/evidential-deep-learning-for-uncertainty","slug":"evidential-deep-learning-for-uncertainty","title":"Evidential Deep Learning for Uncertainty Quantification and Out-of-Distribution Detection in Jet Identification using Deep Neural Networks","date":"2025-01-10","arxiv_id":"2501.05656","repositories_listed":1,"syntology":null},{"url":"/paper/ovo-bench-how-far-is-your-video-llms-from","slug":"ovo-bench-how-far-is-your-video-llms-from","title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","date":"2025-01-09","arxiv_id":"2501.05510","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ovo-bench-how-far-is-your-video-llms-from#ran","syntology_url":"https://syntology.ai/paper/2501.05510","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05510"}},"official":{"repos":["joeleelyf/ovo-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voxeval-benchmarking-the-knowledge","slug":"voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","arxiv_id":"2501.04962","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxeval-benchmarking-the-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.04962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04962"}},"official":{"repos":["dreamtheater123/voxeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/iolbench-benchmarking-llms-on-linguistic","slug":"iolbench-benchmarking-llms-on-linguistic","title":"IOLBENCH: Benchmarking LLMs on Linguistic Reasoning","date":"2025-01-08","arxiv_id":"2501.04249","repositories_listed":1,"syntology":null},{"url":"/paper/underwater-image-restoration-through-a-prior","slug":"underwater-image-restoration-through-a-prior","title":"Underwater Image Restoration Through a Prior Guided Hybrid Sense Approach and Extensive Benchmark Analysis","date":"2025-01-06","arxiv_id":"2501.02701","repositories_listed":1,"syntology":null},{"url":"/paper/tougher-text-smarter-models-raising-the-bar","slug":"tougher-text-smarter-models-raising-the-bar","title":"Tougher Text, Smarter Models: Raising the Bar for Adversarial Defence Benchmarks","date":"2025-01-05","arxiv_id":"2501.02654","repositories_listed":1,"syntology":null},{"url":"/paper/boxinggym-benchmarking-progress-in-automated","slug":"boxinggym-benchmarking-progress-in-automated","title":"BoxingGym: Benchmarking Progress in Automated Experimental Design and Model Discovery","date":"2025-01-02","arxiv_id":"2501.01540","repositories_listed":1,"syntology":null},{"url":"/paper/cysecbench-generative-ai-based-cybersecurity","slug":"cysecbench-generative-ai-based-cybersecurity","title":"CySecBench: Generative AI-based CyberSecurity-focused Prompt Dataset for Benchmarking Large Language Models","date":"2025-01-02","arxiv_id":"2501.01335","repositories_listed":1,"syntology":null},{"url":"/paper/nnwnet-rethinking-the-use-of-transformers-in","slug":"nnwnet-rethinking-the-use-of-transformers-in","title":"nnWNet: Rethinking the Use of Transformers in Biomedical Image Segmentation and Calling for a Unified Evaluation Benchmark","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rcp-bench-benchmarking-robustness-for","slug":"rcp-bench-benchmarking-robustness-for","title":"RCP-Bench: Benchmarking Robustness for Collaborative Perception Under Diverse Corruptions","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ocrbench-v2-an-improved-benchmark-for","slug":"ocrbench-v2-an-improved-benchmark-for","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","date":"2024-12-31","arxiv_id":"2501.00321","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocrbench-v2-an-improved-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2501.00321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00321"}},"official":{"repos":["yuliang-liu/multimodalocr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trajlearn-trajectory-prediction-learning","slug":"trajlearn-trajectory-prediction-learning","title":"TrajLearn: Trajectory Prediction Learning using Deep Generative Models","date":"2024-12-30","arxiv_id":"2501.00184","repositories_listed":1,"syntology":null},{"url":"/paper/on-dataset-transferability-in-medical-image","slug":"on-dataset-transferability-in-medical-image","title":"On dataset transferability in medical image classification","date":"2024-12-28","arxiv_id":"2412.20172","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-optimization-of-portfolio-allocation","slug":"dynamic-optimization-of-portfolio-allocation","title":"A Deep Reinforcement Learning Framework for Dynamic Portfolio Optimization: Evidence from China's Stock Market","date":"2024-12-24","arxiv_id":"2412.18563","repositories_listed":1,"syntology":null},{"url":"/paper/mixmas-a-framework-for-sampling-based-mixer","slug":"mixmas-a-framework-for-sampling-based-mixer","title":"MixMAS: A Framework for Sampling-Based Mixer Architecture Search for Multimodal Fusion and Learning","date":"2024-12-24","arxiv_id":"2412.18437","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-generative-ai-models-for-deep","slug":"benchmarking-generative-ai-models-for-deep","title":"Benchmarking Generative AI Models for Deep Learning Test Input Generation","date":"2024-12-23","arxiv_id":"2412.17652","repositories_listed":1,"syntology":null},{"url":"/paper/dora-sampling-and-benchmarking-for-3d-shape-1","slug":"dora-sampling-and-benchmarking-for-3d-shape-1","title":"Dora: Sampling and Benchmarking for 3D Shape Variational Auto-Encoders","date":"2024-12-23","arxiv_id":"2412.17808","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dora-sampling-and-benchmarking-for-3d-shape-1#ran","syntology_url":"https://syntology.ai/paper/2412.17808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.17808"}},"official":{"repos":["Seed3D/Dora"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-generalization-ability-of-machine","slug":"on-the-generalization-ability-of-machine","title":"On the Generalization Ability of Machine-Generated Text Detectors","date":"2024-12-23","arxiv_id":"2412.17242","repositories_listed":1,"syntology":null},{"url":"/paper/smac-hard-enabling-mixed-opponent-strategy","slug":"smac-hard-enabling-mixed-opponent-strategy","title":"SMAC-Hard: Enabling Mixed Opponent Strategy Script and Self-play on SMAC","date":"2024-12-23","arxiv_id":"2412.17707","repositories_listed":1,"syntology":null},{"url":"/paper/an-openmind-for-3d-medical-vision-self","slug":"an-openmind-for-3d-medical-vision-self","title":"An OpenMind for 3D medical vision self-supervised learning","date":"2024-12-22","arxiv_id":"2412.17041","repositories_listed":1,"syntology":null},{"url":"/paper/first-frame-supervised-video-polyp","slug":"first-frame-supervised-video-polyp","title":"First-frame Supervised Video Polyp Segmentation via Propagative and Semantic Dual-teacher Network","date":"2024-12-21","arxiv_id":"2412.16503","repositories_listed":1,"syntology":null},{"url":"/paper/hammerbench-fine-grained-function-calling","slug":"hammerbench-fine-grained-function-calling","title":"HammerBench: Fine-Grained Function-Calling Evaluation in Real Mobile Device Scenarios","date":"2024-12-21","arxiv_id":"2412.16516","repositories_listed":1,"syntology":null},{"url":"/paper/a-classification-benchmark-for-artificial","slug":"a-classification-benchmark-for-artificial","title":"A Classification Benchmark for Artificial Intelligence Detection of Laryngeal Cancer from Patient Voice","date":"2024-12-20","arxiv_id":"2412.16267","repositories_listed":1,"syntology":null},{"url":"/paper/ai-generated-image-quality-assessment-in","slug":"ai-generated-image-quality-assessment-in","title":"AI-generated Image Quality Assessment in Visual Communication","date":"2024-12-20","arxiv_id":"2412.15677","repositories_listed":1,"syntology":null},{"url":"/paper/deciphering-the-underserved-benchmarking-llm","slug":"deciphering-the-underserved-benchmarking-llm","title":"Deciphering the Underserved: Benchmarking LLM OCR for Low-Resource Scripts","date":"2024-12-20","arxiv_id":"2412.16119","repositories_listed":1,"syntology":null},{"url":"/paper/enriching-social-science-research-via-survey","slug":"enriching-social-science-research-via-survey","title":"Enriching Social Science Research via Survey Item Linking","date":"2024-12-20","arxiv_id":"2412.15831","repositories_listed":1,"syntology":null},{"url":"/paper/xrag-examining-the-core-benchmarking","slug":"xrag-examining-the-core-benchmarking","title":"XRAG: eXamining the Core -- Benchmarking Foundational Components in Advanced Retrieval-Augmented Generation","date":"2024-12-20","arxiv_id":"2412.15529","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/xrag-examining-the-core-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2412.15529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15529"}},"official":{"repos":["docailab/xrag"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/autotrust-benchmarking-trustworthiness-in","slug":"autotrust-benchmarking-trustworthiness-in","title":"AutoTrust: Benchmarking Trustworthiness in Large Vision Language Models for Autonomous Driving","date":"2024-12-19","arxiv_id":"2412.15206","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autotrust-benchmarking-trustworthiness-in#ran","syntology_url":"https://syntology.ai/paper/2412.15206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15206"}},"official":{"repos":["taco-group/autotrust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-ckm-construction-using-partially","slug":"generative-ckm-construction-using-partially","title":"Generative CKM Construction using Partially Observed Data with Diffusion Model","date":"2024-12-19","arxiv_id":"2412.14812","repositories_listed":1,"syntology":null},{"url":"/paper/pitfalls-of-topology-aware-image-segmentation","slug":"pitfalls-of-topology-aware-image-segmentation","title":"Pitfalls of topology-aware image segmentation","date":"2024-12-19","arxiv_id":"2412.14619","repositories_listed":1,"syntology":null},{"url":"/paper/tomg-bench-evaluating-llms-on-text-based-open","slug":"tomg-bench-evaluating-llms-on-text-based-open","title":"TOMG-Bench: Evaluating LLMs on Text-based Open Molecule Generation","date":"2024-12-19","arxiv_id":"2412.14642","repositories_listed":1,"syntology":null},{"url":"/paper/antileak-bench-preventing-data-contamination","slug":"antileak-bench-preventing-data-contamination","title":"AntiLeak-Bench: Preventing Data Contamination by Automatically Constructing Benchmarks with Updated Real-World Knowledge","date":"2024-12-18","arxiv_id":"2412.13670","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/antileak-bench-preventing-data-contamination#ran","syntology_url":"https://syntology.ai/paper/2412.13670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13670"}},"official":{"repos":["bobxwu/antileak-bench"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autonomous-microscopy-experiments-through","slug":"autonomous-microscopy-experiments-through","title":"Autonomous Microscopy Experiments through Large Language Model Agents","date":"2024-12-18","arxiv_id":"2501.10385","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-improving-large-vision","slug":"benchmarking-and-improving-large-vision","title":"Benchmarking and Improving Large Vision-Language Models for Fundamental Visual Graph Understanding and Reasoning","date":"2024-12-18","arxiv_id":"2412.13540","repositories_listed":1,"syntology":null},{"url":"/paper/open-universal-arabic-asr-leaderboard","slug":"open-universal-arabic-asr-leaderboard","title":"Open Universal Arabic ASR Leaderboard","date":"2024-12-18","arxiv_id":"2412.13788","repositories_listed":1,"syntology":null},{"url":"/paper/rag-rewardbench-benchmarking-reward-models-in","slug":"rag-rewardbench-benchmarking-reward-models-in","title":"RAG-RewardBench: Benchmarking Reward Models in Retrieval Augmented Generation for Preference Alignment","date":"2024-12-18","arxiv_id":"2412.13746","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-understanding-compositional","slug":"benchmarking-and-understanding-compositional","title":"Benchmarking and Understanding Compositional Relational Reasoning of LLMs","date":"2024-12-17","arxiv_id":"2412.12841","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-and-understanding-compositional#ran","syntology_url":"https://syntology.ai/paper/2412.12841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12841"}},"official":{"repos":["caiyun-ai/gar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/datelogicqa-benchmarking-temporal-biases-in","slug":"datelogicqa-benchmarking-temporal-biases-in","title":"DateLogicQA: Benchmarking Temporal Biases in Large Language Models","date":"2024-12-17","arxiv_id":"2412.13377","repositories_listed":1,"syntology":null},{"url":"/paper/characterbench-benchmarking-character","slug":"characterbench-benchmarking-character","title":"CharacterBench: Benchmarking Character Customization of Large Language Models","date":"2024-12-16","arxiv_id":"2412.11912","repositories_listed":1,"syntology":null},{"url":"/paper/mt-lens-an-all-in-one-toolkit-for-better","slug":"mt-lens-an-all-in-one-toolkit-for-better","title":"MT-LENS: An all-in-one Toolkit for Better Machine Translation Evaluation","date":"2024-12-16","arxiv_id":"2412.11615","repositories_listed":1,"syntology":null},{"url":"/paper/quench-measuring-the-gap-between-indic-and","slug":"quench-measuring-the-gap-between-indic-and","title":"QUENCH: Measuring the gap between Indic and Non-Indic Contextual General Reasoning in LLMs","date":"2024-12-16","arxiv_id":"2412.11763","repositories_listed":1,"syntology":null},{"url":"/paper/scifaultyqa-benchmarking-llms-on-faulty","slug":"scifaultyqa-benchmarking-llms-on-faulty","title":"SciFaultyQA: Benchmarking LLMs on Faulty Science Question Detection with a GAN-Inspired Approach to Synthetic Dataset Generation","date":"2024-12-16","arxiv_id":"2412.11988","repositories_listed":1,"syntology":null},{"url":"/paper/rolargesum-a-large-dialect-aware-romanian","slug":"rolargesum-a-large-dialect-aware-romanian","title":"RoLargeSum: A Large Dialect-Aware Romanian News Dataset for Summary, Headline, and Keyword Generation","date":"2024-12-15","arxiv_id":"2412.11317","repositories_listed":1,"syntology":null},{"url":"/paper/neuralplexer3-physio-realistic-biomolecular","slug":"neuralplexer3-physio-realistic-biomolecular","title":"NeuralPLexer3: Accurate Biomolecular Complex Structure Prediction with Flow Models","date":"2024-12-14","arxiv_id":"2412.10743","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-linguistic-diversity-of-large","slug":"benchmarking-linguistic-diversity-of-large","title":"Benchmarking Linguistic Diversity of Large Language Models","date":"2024-12-13","arxiv_id":"2412.10271","repositories_listed":1,"syntology":null},{"url":"/paper/evalgim-a-library-for-evaluating-generative","slug":"evalgim-a-library-for-evaluating-generative","title":"EvalGIM: A Library for Evaluating Generative Image Models","date":"2024-12-13","arxiv_id":"2412.10604","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evalgim-a-library-for-evaluating-generative#ran","syntology_url":"https://syntology.ai/paper/2412.10604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10604"}},"official":{"repos":["facebookresearch/evalgim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neptune-the-long-orbit-to-benchmarking-long","slug":"neptune-the-long-orbit-to-benchmarking-long","title":"Neptune: The Long Orbit to Benchmarking Long Video Understanding","date":"2024-12-12","arxiv_id":"2412.09582","repositories_listed":1,"syntology":null},{"url":"/paper/a-quantum-classical-reinforcement-learning","slug":"a-quantum-classical-reinforcement-learning","title":"A quantum-classical reinforcement learning model to play Atari games","date":"2024-12-11","arxiv_id":"2412.08725","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-federated-learning-for-semantic","slug":"benchmarking-federated-learning-for-semantic","title":"Benchmarking Federated Learning for Semantic Datasets: Federated Scene Graph Generation","date":"2024-12-11","arxiv_id":"2412.10436","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-vision-language-models-via","slug":"benchmarking-large-vision-language-models-via","title":"Benchmarking Large Vision-Language Models via Directed Scene Graph for Comprehensive Image Captioning","date":"2024-12-11","arxiv_id":"2412.08614","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2412.08614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08614"}},"official":{"repos":["lufan31/comprecap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/illusory-vqa-benchmarking-and-enhancing","slug":"illusory-vqa-benchmarking-and-enhancing","title":"Illusory VQA: Benchmarking and Enhancing Multimodal Models on Visual Illusions","date":"2024-12-11","arxiv_id":"2412.08169","repositories_listed":1,"syntology":null},{"url":"/paper/learn-how-to-query-from-unlabeled-data","slug":"learn-how-to-query-from-unlabeled-data","title":"Learn How to Query from Unlabeled Data Streams in Federated Learning","date":"2024-12-11","arxiv_id":"2412.08138","repositories_listed":1,"syntology":null},{"url":"/paper/bilingual-bsard-extending-statutory-article","slug":"bilingual-bsard-extending-statutory-article","title":"Bilingual BSARD: Extending Statutory Article Retrieval to Dutch","date":"2024-12-10","arxiv_id":"2412.07462","repositories_listed":1,"syntology":null},{"url":"/paper/graph-neural-networks-are-more-than-filters","slug":"graph-neural-networks-are-more-than-filters","title":"Graph Neural Networks Are More Than Filters: Revisiting and Benchmarking from A Spectral Perspective","date":"2024-12-10","arxiv_id":"2412.07188","repositories_listed":1,"syntology":null},{"url":"/paper/omnidocbench-benchmarking-diverse-pdf","slug":"omnidocbench-benchmarking-diverse-pdf","title":"OmniDocBench: Benchmarking Diverse PDF Document Parsing with Comprehensive Annotations","date":"2024-12-10","arxiv_id":"2412.07626","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/omnidocbench-benchmarking-diverse-pdf#ran","syntology_url":"https://syntology.ai/paper/2412.07626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07626"}},"official":{"repos":["opendatalab/OmniDocBench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-behavior-recommendation-with","slug":"multi-behavior-recommendation-with","title":"Multi-Behavior Recommendation with Personalized Directed Acyclic Behavior Graphs","date":"2024-12-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pediabench-a-comprehensive-chinese-pediatric","slug":"pediabench-a-comprehensive-chinese-pediatric","title":"PediaBench: A Comprehensive Chinese Pediatric Dataset for Benchmarking Large Language Models","date":"2024-12-09","arxiv_id":"2412.06287","repositories_listed":1,"syntology":null},{"url":"/paper/powermamba-a-deep-state-space-model-and","slug":"powermamba-a-deep-state-space-model-and","title":"PowerMamba: A Deep State Space Model and Comprehensive Benchmark for Time Series Prediction in Electric Power Systems","date":"2024-12-09","arxiv_id":"2412.06112","repositories_listed":1,"syntology":null},{"url":"/paper/an-experimental-evaluation-of-imputation","slug":"an-experimental-evaluation-of-imputation","title":"An Experimental Evaluation of Imputation Models for Spatial-Temporal Traffic Data","date":"2024-12-06","arxiv_id":"2412.04733","repositories_listed":1,"syntology":null}],"record_sha256":"7c5b77a9124972cffa3654973e0f9a46cf5869eec2dd8ac998dc8c08d1a8b06a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}