{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/6","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":56,"rows_per_page":100,"rows":[501,600],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/5","next":"/task/benchmarking/papers/7","papers":[{"url":"/paper/codemenv-benchmarking-large-language-models","slug":"codemenv-benchmarking-large-language-models","title":"CODEMENV: Benchmarking Large Language Models on Code Migration","date":"2025-06-01","arxiv_id":"2506.00894","repositories_listed":1,"syntology":null},{"url":"/paper/medbookvqa-a-systematic-and-comprehensive","slug":"medbookvqa-a-systematic-and-comprehensive","title":"MedBookVQA: A Systematic and Comprehensive Medical Benchmark Derived from Open-Access Book","date":"2025-06-01","arxiv_id":"2506.00855","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-avrobustbench-benchmarking-the","slug":"texttt-avrobustbench-benchmarking-the","title":"$\\texttt{AVROBUSTBENCH}$: Benchmarking the Robustness of Audio-Visual Recognition Models at Test-Time","date":"2025-05-31","arxiv_id":"2506.00358","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/texttt-avrobustbench-benchmarking-the#ran","syntology_url":"https://syntology.ai/paper/2506.00358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00358"}},"official":{"repos":["sarthaxxxxx/av-c-robustness-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bench4ke-benchmarking-automated-competency","slug":"bench4ke-benchmarking-automated-competency","title":"Bench4KE: Benchmarking Automated Competency Question Generation","date":"2025-05-30","arxiv_id":"2505.24554","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-atomic-geometry-representations-in","slug":"beyond-atomic-geometry-representations-in","title":"Beyond Atomic Geometry Representations in Materials Science: A Human-in-the-Loop Multimodal Framework","date":"2025-05-30","arxiv_id":"2506.00302","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/beyond-atomic-geometry-representations-in#ran","syntology_url":"https://syntology.ai/paper/2506.00302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00302"}},"official":{"repos":["kurbanintelligencelab/multicrystalspectrumset"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/byzfl-research-framework-for-robust-federated","slug":"byzfl-research-framework-for-robust-federated","title":"ByzFL: Research Framework for Robust Federated Learning","date":"2025-05-30","arxiv_id":"2505.24802","repositories_listed":1,"syntology":null},{"url":"/paper/draw-all-your-imagine-a-holistic-benchmark","slug":"draw-all-your-imagine-a-holistic-benchmark","title":"Draw ALL Your Imagine: A Holistic Benchmark and Agent Framework for Complex Instruction-based Image Generation","date":"2025-05-30","arxiv_id":"2505.24787","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/draw-all-your-imagine-a-holistic-benchmark#ran","syntology_url":"https://syntology.ai/paper/2505.24787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24787"}},"official":{"repos":["yczhou001/longbench-t2i"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/metafaith-faithful-natural-language","slug":"metafaith-faithful-natural-language","title":"MetaFaith: Faithful Natural Language Uncertainty Expression in LLMs","date":"2025-05-30","arxiv_id":"2505.24858","repositories_listed":1,"syntology":null},{"url":"/paper/open-captchaworld-a-comprehensive-web-based","slug":"open-captchaworld-a-comprehensive-web-based","title":"Open CaptchaWorld: A Comprehensive Web-based Platform for Testing and Benchmarking Multimodal LLM Agents","date":"2025-05-30","arxiv_id":"2505.24878","repositories_listed":1,"syntology":null},{"url":"/paper/pathgene-benchmarking-driver-gene-mutations","slug":"pathgene-benchmarking-driver-gene-mutations","title":"PathGene: Benchmarking Driver Gene Mutations and Exon Prediction Using Multicenter Lung Cancer Histopathology Image Dataset","date":"2025-05-30","arxiv_id":"2506.00096","repositories_listed":1,"syntology":null},{"url":"/paper/segmenting-france-across-four-centuries","slug":"segmenting-france-across-four-centuries","title":"Segmenting France Across Four Centuries","date":"2025-05-30","arxiv_id":"2505.24824","repositories_listed":1,"syntology":null},{"url":"/paper/sorce-small-object-retrieval-in-complex","slug":"sorce-small-object-retrieval-in-complex","title":"SORCE: Small Object Retrieval in Complex Environments","date":"2025-05-30","arxiv_id":"2505.24441","repositories_listed":1,"syntology":null},{"url":"/paper/is-your-model-fairly-certain-uncertainty","slug":"is-your-model-fairly-certain-uncertainty","title":"Is Your Model Fairly Certain? Uncertainty-Aware Fairness Evaluation for LLMs","date":"2025-05-29","arxiv_id":"2505.23996","repositories_listed":1,"syntology":null},{"url":"/paper/llm-performance-for-code-generation-on-noisy","slug":"llm-performance-for-code-generation-on-noisy","title":"LLM Performance for Code Generation on Noisy Tasks","date":"2025-05-29","arxiv_id":"2505.23598","repositories_listed":1,"syntology":null},{"url":"/paper/sns-bench-vl-benchmarking-multimodal-large","slug":"sns-bench-vl-benchmarking-multimodal-large","title":"SNS-Bench-VL: Benchmarking Multimodal Large Language Models in Social Networking Services","date":"2025-05-29","arxiv_id":"2505.23065","repositories_listed":1,"syntology":null},{"url":"/paper/toward-memory-aided-world-models-benchmarking","slug":"toward-memory-aided-world-models-benchmarking","title":"Toward Memory-Aided World Models: Benchmarking via Spatial Consistency","date":"2025-05-29","arxiv_id":"2505.22976","repositories_listed":1,"syntology":null},{"url":"/paper/verina-benchmarking-verifiable-code","slug":"verina-benchmarking-verifiable-code","title":"VERINA: Benchmarking Verifiable Code Generation","date":"2025-05-29","arxiv_id":"2505.23135","repositories_listed":1,"syntology":null},{"url":"/paper/b-xaic-dataset-benchmarking-explainable-ai","slug":"b-xaic-dataset-benchmarking-explainable-ai","title":"B-XAIC Dataset: Benchmarking Explainable AI for Graph Neural Networks Using Chemical Data","date":"2025-05-28","arxiv_id":"2505.22252","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-abstract-and-reasoning-abilities","slug":"benchmarking-abstract-and-reasoning-abilities","title":"Benchmarking Abstract and Reasoning Abilities Through A Theoretical Perspective","date":"2025-05-28","arxiv_id":"2505.23833","repositories_listed":1,"syntology":null},{"url":"/paper/characterizing-bias-benchmarking-large","slug":"characterizing-bias-benchmarking-large","title":"Characterizing Bias: Benchmarking Large Language Models in Simplified versus Traditional Chinese","date":"2025-05-28","arxiv_id":"2505.22645","repositories_listed":1,"syntology":null},{"url":"/paper/gomatching-parameter-and-data-efficient","slug":"gomatching-parameter-and-data-efficient","title":"GoMatching++: Parameter- and Data-Efficient Arbitrary-Shaped Video Text Spotting and Benchmarking","date":"2025-05-28","arxiv_id":"2505.22228","repositories_listed":1,"syntology":null},{"url":"/paper/medal-a-framework-for-benchmarking-llms-as","slug":"medal-a-framework-for-benchmarking-llms-as","title":"MEDAL: A Framework for Benchmarking LLMs as Multilingual Open-Domain Chatbots and Dialogue Evaluators","date":"2025-05-28","arxiv_id":"2505.22777","repositories_listed":1,"syntology":null},{"url":"/paper/redteamcua-realistic-adversarial-testing-of","slug":"redteamcua-realistic-adversarial-testing-of","title":"RedTeamCUA: Realistic Adversarial Testing of Computer-Use Agents in Hybrid Web-OS Environments","date":"2025-05-28","arxiv_id":"2505.21936","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/redteamcua-realistic-adversarial-testing-of#ran","syntology_url":"https://syntology.ai/paper/2505.21936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21936"}},"official":{"repos":["osu-nlp-group/redteamcua"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-parameter-and-memory-efficient","slug":"scalable-parameter-and-memory-efficient","title":"Scalable Parameter and Memory Efficient Pretraining for LLM: Recent Algorithmic Advances and Benchmarking","date":"2025-05-28","arxiv_id":"2505.22922","repositories_listed":1,"syntology":null},{"url":"/paper/starbase-gp-biologically-guided-automated","slug":"starbase-gp-biologically-guided-automated","title":"StarBASE-GP: Biologically-Guided Automated Machine Learning for Genotype-to-Phenotype Association Analysis","date":"2025-05-28","arxiv_id":"2505.22746","repositories_listed":1,"syntology":null},{"url":"/paper/svrpbench-a-realistic-benchmark-for","slug":"svrpbench-a-realistic-benchmark-for","title":"SVRPBench: A Realistic Benchmark for Stochastic Vehicle Routing Problem","date":"2025-05-28","arxiv_id":"2505.21887","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/svrpbench-a-realistic-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2505.21887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21887"}},"official":{"repos":["yehias21/vrp-benchmarks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autojudger-an-agent-driven-framework-for","slug":"autojudger-an-agent-driven-framework-for","title":"AutoJudger: An Agent-Driven Framework for Efficient Benchmarking of MLLMs","date":"2025-05-27","arxiv_id":"2505.21389","repositories_listed":1,"syntology":null},{"url":"/paper/bencher-simple-and-reproducible-benchmarking","slug":"bencher-simple-and-reproducible-benchmarking","title":"Bencher: Simple and Reproducible Benchmarking for Black-Box Optimization","date":"2025-05-27","arxiv_id":"2505.21321","repositories_listed":1,"syntology":null},{"url":"/paper/fedivertex-a-graph-dataset-based-on","slug":"fedivertex-a-graph-dataset-based-on","title":"Fedivertex: a Graph Dataset based on Decentralized Social Networks for Trustworthy Machine Learning","date":"2025-05-27","arxiv_id":"2505.20882","repositories_listed":1,"syntology":null},{"url":"/paper/fm-planner-foundation-model-guided-path","slug":"fm-planner-foundation-model-guided-path","title":"FM-Planner: Foundation Model Guided Path Planning for Autonomous Drone Navigation","date":"2025-05-27","arxiv_id":"2505.20783","repositories_listed":1,"syntology":null},{"url":"/paper/frames-vqa-benchmarking-fine-tuning-1","slug":"frames-vqa-benchmarking-fine-tuning-1","title":"FRAMES-VQA: Benchmarking Fine-Tuning Robustness across Multi-Modal Shifts in Visual Question Answering","date":"2025-05-27","arxiv_id":"2505.21755","repositories_listed":1,"syntology":null},{"url":"/paper/laparoscopic-image-desmoking-using-the-u-net","slug":"laparoscopic-image-desmoking-using-the-u-net","title":"Laparoscopic Image Desmoking Using the U-Net with New Loss Function and Integrated Differentiable Wiener Filter","date":"2025-05-27","arxiv_id":"2505.21634","repositories_listed":1,"syntology":null},{"url":"/paper/videomarkbench-benchmarking-robustness-of","slug":"videomarkbench-benchmarking-robustness-of","title":"VideoMarkBench: Benchmarking Robustness of Video Watermarking","date":"2025-05-27","arxiv_id":"2505.21620","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomarkbench-benchmarking-robustness-of#ran","syntology_url":"https://syntology.ai/paper/2505.21620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21620"}},"official":{"repos":["zhengyuan-jiang/videomarkbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/amqa-an-adversarial-dataset-for-benchmarking","slug":"amqa-an-adversarial-dataset-for-benchmarking","title":"AMQA: An Adversarial Dataset for Benchmarking Bias of LLMs in Medicine and Healthcare","date":"2025-05-26","arxiv_id":"2505.19562","repositories_listed":1,"syntology":null},{"url":"/paper/automated-text-to-table-for-reasoning","slug":"automated-text-to-table-for-reasoning","title":"Automated Text-to-Table for Reasoning-Intensive Table QA: Pipeline Design and Benchmarking Insights","date":"2025-05-26","arxiv_id":"2505.19563","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-enhancing-llm-agents-in","slug":"benchmarking-and-enhancing-llm-agents-in","title":"Benchmarking and Enhancing LLM Agents in Localizing Linux Kernel Bugs","date":"2025-05-26","arxiv_id":"2505.19489","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-knowledge-conflict","slug":"benchmarking-multimodal-knowledge-conflict","title":"Benchmarking Multimodal Knowledge Conflict for Large Multimodal Models","date":"2025-05-26","arxiv_id":"2505.19509","repositories_listed":1,"syntology":null},{"url":"/paper/calibrating-pre-trained-language-classifiers","slug":"calibrating-pre-trained-language-classifiers","title":"Calibrating Pre-trained Language Classifiers on LLM-generated Noisy Labels via Iterative Refinement","date":"2025-05-26","arxiv_id":"2505.19675","repositories_listed":1,"syntology":null},{"url":"/paper/mineanybuild-benchmarking-spatial-planning","slug":"mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","arxiv_id":"2505.20148","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineanybuild-benchmarking-spatial-planning#ran","syntology_url":"https://syntology.ai/paper/2505.20148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20148"}},"official":{"repos":["mineanybuild/mineanybuild"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/ob3d-a-new-dataset-for-benchmarking","slug":"ob3d-a-new-dataset-for-benchmarking","title":"OB3D: A New Dataset for Benchmarking Omnidirectional 3D Reconstruction Using Blender","date":"2025-05-26","arxiv_id":"2505.20126","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-time-series-forecasting-with","slug":"synthetic-time-series-forecasting-with","title":"Synthetic Time Series Forecasting with Transformer Architectures: Extensive Simulation Benchmarks","date":"2025-05-26","arxiv_id":"2505.20048","repositories_listed":1,"syntology":null},{"url":"/paper/are-vision-language-models-ready-for-clinical","slug":"are-vision-language-models-ready-for-clinical","title":"Are Vision Language Models Ready for Clinical Diagnosis? A 3D Medical Benchmark for Tumor-centric Visual Question Answering","date":"2025-05-25","arxiv_id":"2505.18915","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vision-language-models-ready-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2505.18915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18915"}},"official":{"repos":["schuture/deeptumorvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-laparoscopic-surgical-image","slug":"benchmarking-laparoscopic-surgical-image","title":"Benchmarking Laparoscopic Surgical Image Restoration and Beyond","date":"2025-05-25","arxiv_id":"2505.19161","repositories_listed":1,"syntology":null},{"url":"/paper/seephys-does-seeing-help-thinking","slug":"seephys-does-seeing-help-thinking","title":"SeePhys: Does Seeing Help Thinking? -- Benchmarking Vision-Based Physics Reasoning","date":"2025-05-25","arxiv_id":"2505.19099","repositories_listed":1,"syntology":null},{"url":"/paper/crmarena-pro-holistic-assessment-of-llm","slug":"crmarena-pro-holistic-assessment-of-llm","title":"CRMArena-Pro: Holistic Assessment of LLM Agents Across Diverse Business Scenarios and Interactions","date":"2025-05-24","arxiv_id":"2505.18878","repositories_listed":1,"syntology":null},{"url":"/paper/spdebench-an-extensive-benchmark-for-learning","slug":"spdebench-an-extensive-benchmark-for-learning","title":"SPDEBench: An Extensive Benchmark for Learning Regular and Singular Stochastic PDEs","date":"2025-05-24","arxiv_id":"2505.18511","repositories_listed":1,"syntology":null},{"url":"/paper/3d-face-reconstruction-error-decomposed-a","slug":"3d-face-reconstruction-error-decomposed-a","title":"3D Face Reconstruction Error Decomposed: A Modular Benchmark for Fair and Fast Method Evaluation","date":"2025-05-23","arxiv_id":"2505.18025","repositories_listed":1,"syntology":null},{"url":"/paper/a-position-paper-on-the-automatic-generation","slug":"a-position-paper-on-the-automatic-generation","title":"A Position Paper on the Automatic Generation of Machine Learning Leaderboards","date":"2025-05-23","arxiv_id":"2505.17465","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-recommendation-classification","slug":"benchmarking-recommendation-classification","title":"Benchmarking Recommendation, Classification, and Tracing Based on Hugging Face Knowledge Graph","date":"2025-05-23","arxiv_id":"2505.17507","repositories_listed":1,"syntology":null},{"url":"/paper/fullfront-benchmarking-mllms-across-the-full","slug":"fullfront-benchmarking-mllms-across-the-full","title":"FullFront: Benchmarking MLLMs Across the Full Front-End Engineering Workflow","date":"2025-05-23","arxiv_id":"2505.17399","repositories_listed":1,"syntology":null},{"url":"/paper/jalmbench-benchmarking-jailbreak","slug":"jalmbench-benchmarking-jailbreak","title":"JALMBench: Benchmarking Jailbreak Vulnerabilities in Audio Language Models","date":"2025-05-23","arxiv_id":"2505.17568","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-correspondence-unified-benchmarking","slug":"semantic-correspondence-unified-benchmarking","title":"Semantic Correspondence: Unified Benchmarking and a Strong Baseline","date":"2025-05-23","arxiv_id":"2505.18060","repositories_listed":1,"syntology":null},{"url":"/paper/semsegbench-detecbench-benchmarking","slug":"semsegbench-detecbench-benchmarking","title":"SemSegBench & DetecBench: Benchmarking Reliability and Generalization Beyond Classification","date":"2025-05-23","arxiv_id":"2505.18015","repositories_listed":1,"syntology":null},{"url":"/paper/twin-2k-500-a-dataset-for-building-digital","slug":"twin-2k-500-a-dataset-for-building-digital","title":"Twin-2K-500: A dataset for building digital twins of over 2,000 people based on their answers to over 500 questions","date":"2025-05-23","arxiv_id":"2505.17479","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/twin-2k-500-a-dataset-for-building-digital#ran","syntology_url":"https://syntology.ai/paper/2505.17479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17479"}},"official":{"repos":["tianyipeng-lab/digital-twin-simulation"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/wildfire-spread-forecasting-with-deep","slug":"wildfire-spread-forecasting-with-deep","title":"Wildfire spread forecasting with Deep Learning","date":"2025-05-23","arxiv_id":"2505.17556","repositories_listed":1,"syntology":null},{"url":"/paper/agentif-benchmarking-instruction-following-of","slug":"agentif-benchmarking-instruction-following-of","title":"AGENTIF: Benchmarking Instruction Following of Large Language Models in Agentic Scenarios","date":"2025-05-22","arxiv_id":"2505.16944","repositories_listed":1,"syntology":null},{"url":"/paper/audiotrust-benchmarking-the-multifaceted","slug":"audiotrust-benchmarking-the-multifaceted","title":"AudioTrust: Benchmarking the Multifaceted Trustworthiness of Audio Large Language Models","date":"2025-05-22","arxiv_id":"2505.16211","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiotrust-benchmarking-the-multifaceted#ran","syntology_url":"https://syntology.ai/paper/2505.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16211"}},"official":{"repos":["jusperlee/audiotrust"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-retrieval-augmented-multimomal","slug":"benchmarking-retrieval-augmented-multimomal","title":"Benchmarking Retrieval-Augmented Multimomal Generation for Document Question Answering","date":"2025-05-22","arxiv_id":"2505.16470","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-multimomal#ran","syntology_url":"https://syntology.ai/paper/2505.16470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16470"}},"official":{"repos":["mmdocrag/mmdocrag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/edubench-a-comprehensive-benchmarking-dataset","slug":"edubench-a-comprehensive-benchmarking-dataset","title":"EduBench: A Comprehensive Benchmarking Dataset for Evaluating Large Language Models in Diverse Educational Scenarios","date":"2025-05-22","arxiv_id":"2505.16160","repositories_listed":1,"syntology":null},{"url":"/paper/ifeval-audio-benchmarking-instruction","slug":"ifeval-audio-benchmarking-instruction","title":"IFEval-Audio: Benchmarking Instruction-Following Capability in Audio-based Large Language Models","date":"2025-05-22","arxiv_id":"2505.16774","repositories_listed":1,"syntology":null},{"url":"/paper/learning-collective-multi-cellular-dynamics","slug":"learning-collective-multi-cellular-dynamics","title":"Learning collective multi-cellular dynamics from temporal scRNA-seq via a transformer-enhanced Neural SDE","date":"2025-05-22","arxiv_id":"2505.16492","repositories_listed":1,"syntology":null},{"url":"/paper/reobench-benchmarking-robustness-of-earth","slug":"reobench-benchmarking-robustness-of-earth","title":"REOBench: Benchmarking Robustness of Earth Observation Foundation Models","date":"2025-05-22","arxiv_id":"2505.16793","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-hyperspectral-pansharpening-using","slug":"zero-shot-hyperspectral-pansharpening-using","title":"Zero-Shot Hyperspectral Pansharpening Using Hysteresis-Based Tuning for Spectral Quality Control","date":"2025-05-22","arxiv_id":"2505.16658","repositories_listed":1,"syntology":null},{"url":"/paper/keep-security-benchmarking-security-policy","slug":"keep-security-benchmarking-security-policy","title":"Keep Security! Benchmarking Security Policy Preservation in Large Language Model Contexts Against Indirect Attacks in Question Answering","date":"2025-05-21","arxiv_id":"2505.15805","repositories_listed":1,"syntology":null},{"url":"/paper/lost-in-benchmarks-rethinking-large-language","slug":"lost-in-benchmarks-rethinking-large-language","title":"Lost in Benchmarks? Rethinking Large Language Model Benchmarking with Item Response Theory","date":"2025-05-21","arxiv_id":"2505.15055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lost-in-benchmarks-rethinking-large-language#ran","syntology_url":"https://syntology.ai/paper/2505.15055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15055"}},"official":{"repos":["Joe-Hall-Lee/PSN-IRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/oral-imaging-for-malocclusion-issues","slug":"oral-imaging-for-malocclusion-issues","title":"Oral Imaging for Malocclusion Issues Assessments: OMNI Dataset, Deep Learning Baselines and Benchmarking","date":"2025-05-21","arxiv_id":"2505.15637","repositories_listed":1,"syntology":null},{"url":"/paper/traveling-across-languages-benchmarking-cross","slug":"traveling-across-languages-benchmarking-cross","title":"Traveling Across Languages: Benchmarking Cross-Lingual Consistency in Multimodal LLMs","date":"2025-05-21","arxiv_id":"2505.15075","repositories_listed":1,"syntology":null},{"url":"/paper/urdufactcheck-an-agentic-fact-checking","slug":"urdufactcheck-an-agentic-fact-checking","title":"UrduFactCheck: An Agentic Fact-Checking Framework for Urdu with Evidence Boosting and Benchmarking","date":"2025-05-21","arxiv_id":"2505.15063","repositories_listed":1,"syntology":null},{"url":"/paper/vocalbench-benchmarking-the-vocal","slug":"vocalbench-benchmarking-the-vocal","title":"VocalBench: Benchmarking the Vocal Conversational Abilities for Speech Interaction Models","date":"2025-05-21","arxiv_id":"2505.15727","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-the-myopic-trap-positional-bias","slug":"benchmarking-the-myopic-trap-positional-bias","title":"Benchmarking the Myopic Trap: Positional Bias in Information Retrieval","date":"2025-05-20","arxiv_id":"2505.13950","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosisarena-benchmarking-diagnostic","slug":"diagnosisarena-benchmarking-diagnostic","title":"DiagnosisArena: Benchmarking Diagnostic Reasoning for Large Language Models","date":"2025-05-20","arxiv_id":"2505.14107","repositories_listed":1,"syntology":null},{"url":"/paper/survunc-a-meta-model-based-uncertainty","slug":"survunc-a-meta-model-based-uncertainty","title":"SurvUnc: A Meta-Model Based Uncertainty Quantification Framework for Survival Analysis","date":"2025-05-20","arxiv_id":"2505.14803","repositories_listed":1,"syntology":null},{"url":"/paper/txpert-leveraging-biochemical-relationships","slug":"txpert-leveraging-biochemical-relationships","title":"TxPert: Leveraging Biochemical Relationships for Out-of-Distribution Transcriptomic Perturbation Prediction","date":"2025-05-20","arxiv_id":"2505.14919","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-confidence-evaluation-of","slug":"benchmarking-and-confidence-evaluation-of","title":"Benchmarking and Confidence Evaluation of LALMs For Temporal Reasoning","date":"2025-05-19","arxiv_id":"2505.13115","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-moeas-for-solving-continuous","slug":"benchmarking-moeas-for-solving-continuous","title":"Benchmarking MOEAs for solving continuous multi-objective RL problems","date":"2025-05-19","arxiv_id":"2505.13726","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-arena-towards-democratic-and","slug":"decentralized-arena-towards-democratic-and","title":"Decentralized Arena: Towards Democratic and Scalable Automatic Evaluation of Language Models","date":"2025-05-19","arxiv_id":"2505.12808","repositories_listed":1,"syntology":null},{"url":"/paper/hr-vilage-3k3m-a-human-respiratory-viral","slug":"hr-vilage-3k3m-a-human-respiratory-viral","title":"HR-VILAGE-3K3M: A Human Respiratory Viral Immunization Longitudinal Gene Expression Dataset for Systems Immunity","date":"2025-05-19","arxiv_id":"2505.14725","repositories_listed":1,"syntology":null},{"url":"/paper/ineq-comp-benchmarking-human-intuitive","slug":"ineq-comp-benchmarking-human-intuitive","title":"Ineq-Comp: Benchmarking Human-Intuitive Compositional Reasoning in Automated Theorem Proving on Inequalities","date":"2025-05-19","arxiv_id":"2505.12680","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ineq-comp-benchmarking-human-intuitive#ran","syntology_url":"https://syntology.ai/paper/2505.12680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12680"}},"official":{"repos":["haoyuzhao123/leanineqcomp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timeseriesgym-a-scalable-benchmark-for-time","slug":"timeseriesgym-a-scalable-benchmark-for-time","title":"TimeSeriesGym: A Scalable Benchmark for (Time Series) Machine Learning Engineering Agents","date":"2025-05-19","arxiv_id":"2505.13291","repositories_listed":1,"syntology":null},{"url":"/paper/globalgeotree-a-multi-granular-vision","slug":"globalgeotree-a-multi-granular-vision","title":"GlobalGeoTree: A Multi-Granular Vision-Language Dataset for Global Tree Species Classification","date":"2025-05-18","arxiv_id":"2505.12513","repositories_listed":1,"syntology":null},{"url":"/paper/medagentboard-benchmarking-multi-agent","slug":"medagentboard-benchmarking-multi-agent","title":"MedAgentBoard: Benchmarking Multi-Agent Collaboration with Conventional Methods for Diverse Medical Tasks","date":"2025-05-18","arxiv_id":"2505.12371","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentboard-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.12371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12371"}},"official":{"repos":["yhzhu99/medagentboard"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/oss-bench-benchmark-generator-for-coding-llms","slug":"oss-bench-benchmark-generator-for-coding-llms","title":"OSS-Bench: Benchmark Generator for Coding LLMs","date":"2025-05-18","arxiv_id":"2505.12331","repositories_listed":1,"syntology":null},{"url":"/paper/what-are-they-talking-about-benchmarking","slug":"what-are-they-talking-about-benchmarking","title":"What are they talking about? Benchmarking Large Language Models for Knowledge-Grounded Discussion Summarization","date":"2025-05-18","arxiv_id":"2505.12474","repositories_listed":1,"syntology":null},{"url":"/paper/genderbench-evaluation-suite-for-gender","slug":"genderbench-evaluation-suite-for-gender","title":"GenderBench: Evaluation Suite for Gender Biases in LLMs","date":"2025-05-17","arxiv_id":"2505.12054","repositories_listed":1,"syntology":null},{"url":"/paper/love-benchmarking-and-evaluating-text-to","slug":"love-benchmarking-and-evaluating-text-to","title":"LOVE: Benchmarking and Evaluating Text-to-Video Generation and Video-to-Text Interpretation","date":"2025-05-17","arxiv_id":"2505.12098","repositories_listed":1,"syntology":null},{"url":"/paper/softpq-robust-instance-segmentation","slug":"softpq-robust-instance-segmentation","title":"SoftPQ: Robust Instance Segmentation Evaluation via Soft Matching and Tunable Thresholds","date":"2025-05-17","arxiv_id":"2505.12155","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10852","slug":"2505-10852","title":"MatTools: Benchmarking Large Language Models for Materials Science Tools","date":"2025-05-16","arxiv_id":"2505.10852","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10931","slug":"2505-10931","title":"M4-SAR: A Multi-Resolution, Multi-Polarization, Multi-Scene, Multi-Source Dataset and Benchmark for Optical-SAR Fusion Object Detection","date":"2025-05-16","arxiv_id":"2505.10931","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11034","slug":"2505-11034","title":"CleanPatrick: A Benchmark for Image Data Cleaning","date":"2025-05-16","arxiv_id":"2505.11034","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11151","slug":"2505-11151","title":"STEP: A Unified Spiking Transformer Evaluation Platform for Fair and Reproducible Benchmarking","date":"2025-05-16","arxiv_id":"2505.11151","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11185","slug":"2505-11185","title":"VitaGraph: Building a Knowledge Graph for Biologically Relevant Learning Tasks","date":"2025-05-16","arxiv_id":"2505.11185","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11239","slug":"2505-11239","title":"Massive-STEPS: Massive Semantic Trajectories for Understanding POI Check-ins -- Dataset and Benchmarks","date":"2025-05-16","arxiv_id":"2505.11239","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11454","slug":"2505-11454","title":"HumaniBench: A Human-Centric Framework for Large Multimodal Models Evaluation","date":"2025-05-16","arxiv_id":"2505.11454","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-spatiotemporal-reasoning-in-llms","slug":"benchmarking-spatiotemporal-reasoning-in-llms","title":"Benchmarking Spatiotemporal Reasoning in LLMs and Reasoning Models: Capabilities and Challenges","date":"2025-05-16","arxiv_id":"2505.11618","repositories_listed":1,"syntology":null},{"url":"/paper/tcc-bench-benchmarking-the-traditional","slug":"tcc-bench-benchmarking-the-traditional","title":"TCC-Bench: Benchmarking the Traditional Chinese Culture Understanding Capabilities of MLLMs","date":"2025-05-16","arxiv_id":"2505.11275","repositories_listed":1,"syntology":null},{"url":"/paper/time-travel-is-cheating-going-live-with","slug":"time-travel-is-cheating-going-live-with","title":"Time Travel is Cheating: Going Live with DeepFund for Real-Time Fund Investment Benchmarking","date":"2025-05-16","arxiv_id":"2505.11065","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/time-travel-is-cheating-going-live-with#ran","syntology_url":"https://syntology.ai/paper/2505.11065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11065"}},"official":{"repos":["hkustdial/deepfund"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10610","slug":"2505-10610","title":"MMLongBench: Benchmarking Long-Context Vision-Language Models Effectively and Thoroughly","date":"2025-05-15","arxiv_id":"2505.10610","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10610#ran","syntology_url":"https://syntology.ai/paper/2505.10610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10610"}},"official":{"repos":["edinburghnlp/mmlongbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10711","slug":"2505-10711","title":"GNN-Suite: a Graph Neural Network Benchmarking Framework for Biomedical Informatics","date":"2025-05-15","arxiv_id":"2505.10711","repositories_listed":1,"syntology":null},{"url":"/paper/do-llms-memorize-recommendation-datasets-a","slug":"do-llms-memorize-recommendation-datasets-a","title":"Do LLMs Memorize Recommendation Datasets? A Preliminary Study on MovieLens-1M","date":"2025-05-15","arxiv_id":"2505.10212","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-robustness-of-deep-reinforcement","slug":"evaluating-robustness-of-deep-reinforcement","title":"Evaluating Robustness of Deep Reinforcement Learning for Autonomous Surface Vehicle Control in Field Tests","date":"2025-05-15","arxiv_id":"2505.10033","repositories_listed":1,"syntology":null}],"record_sha256":"511af67cf2f00884e72d6bbb291309a070454505ffe6dbe25847b7e4f4921c60","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}