{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/8","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":56,"rows_per_page":100,"rows":[701,800],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/7","next":"/task/benchmarking/papers/9","papers":[{"url":"/paper/mind-the-gap-benchmarking-spatial-reasoning","slug":"mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19707","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-gap-benchmarking-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19707"}},"official":{"repos":["stogiannidis/srbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neorl-2-near-real-world-benchmarks-for","slug":"neorl-2-near-real-world-benchmarks-for","title":"NeoRL-2: Near Real-World Benchmarks for Offline Reinforcement Learning with Extended Realistic Scenarios","date":"2025-03-25","arxiv_id":"2503.19267","repositories_listed":1,"syntology":null},{"url":"/paper/the-coralscapes-dataset-semantic-scene","slug":"the-coralscapes-dataset-semantic-scene","title":"The Coralscapes Dataset: Semantic Scene Understanding in Coral Reefs","date":"2025-03-25","arxiv_id":"2503.20000","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multi-modal-semantic","slug":"benchmarking-multi-modal-semantic","title":"Benchmarking Multi-modal Semantic Segmentation under Sensor Failures: Missing and Noisy Modality Robustness","date":"2025-03-24","arxiv_id":"2503.18445","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-object-detectors-under-real-1","slug":"benchmarking-object-detectors-under-real-1","title":"Benchmarking Object Detectors under Real-World Distribution Shifts in Satellite Imagery","date":"2025-03-24","arxiv_id":"2503.19202","repositories_listed":1,"syntology":null},{"url":"/paper/llm-benchmarking-with-llama2-evaluating-code","slug":"llm-benchmarking-with-llama2-evaluating-code","title":"LLM Benchmarking with LLaMA2: Evaluating Code Development Performance Across Multiple Programming Languages","date":"2025-03-24","arxiv_id":"2503.19217","repositories_listed":1,"syntology":null},{"url":"/paper/mining-gym-a-configurable-rl-benchmarking","slug":"mining-gym-a-configurable-rl-benchmarking","title":"Mining-Gym: A Configurable RL Benchmarking Environment for Truck Dispatch Scheduling","date":"2025-03-24","arxiv_id":"2503.19195","repositories_listed":1,"syntology":null},{"url":"/paper/accurate-peak-detection-in-multimodal","slug":"accurate-peak-detection-in-multimodal","title":"Accurate Peak Detection in Multimodal Optimization via Approximated Landscape Learning","date":"2025-03-23","arxiv_id":"2503.18066","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/accurate-peak-detection-in-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.18066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18066"}},"official":{"repos":["gmc-drl/apdmmo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/geobenchx-benchmarking-llms-for-multistep","slug":"geobenchx-benchmarking-llms-for-multistep","title":"GeoBenchX: Benchmarking LLMs for Multistep Geospatial Tasks","date":"2025-03-23","arxiv_id":"2503.18129","repositories_listed":1,"syntology":null},{"url":"/paper/scenesplat-gaussian-splatting-based-scene","slug":"scenesplat-gaussian-splatting-based-scene","title":"SceneSplat: Gaussian Splatting-based Scene Understanding with Vision-Language Pretraining","date":"2025-03-23","arxiv_id":"2503.18052","repositories_listed":1,"syntology":null},{"url":"/paper/4d-bench-benchmarking-multi-modal-large","slug":"4d-bench-benchmarking-multi-modal-large","title":"4D-Bench: Benchmarking Multi-modal Large Language Models for 4D Object Understanding","date":"2025-03-22","arxiv_id":"2503.17827","repositories_listed":1,"syntology":null},{"url":"/paper/v2p-bench-evaluating-video-language","slug":"v2p-bench-evaluating-video-language","title":"V2P-Bench: Evaluating Video-Language Understanding with Visual Prompts for Better Human-Model Interaction","date":"2025-03-22","arxiv_id":"2503.17736","repositories_listed":1,"syntology":null},{"url":"/paper/decouple-and-track-benchmarking-and-improving","slug":"decouple-and-track-benchmarking-and-improving","title":"Decouple and Track: Benchmarking and Improving Video Diffusion Transformers for Motion Transfer","date":"2025-03-21","arxiv_id":"2503.17350","repositories_listed":1,"syntology":null},{"url":"/paper/contextgnn-goes-to-elliot-towards","slug":"contextgnn-goes-to-elliot-towards","title":"ContextGNN goes to Elliot: Towards Benchmarking Relational Deep Learning for Static Link Prediction (aka Personalized Item Recommendation)","date":"2025-03-20","arxiv_id":"2503.16661","repositories_listed":1,"syntology":null},{"url":"/paper/qcpinn-quantum-classical-physics-informed","slug":"qcpinn-quantum-classical-physics-informed","title":"QCPINN: Quantum-Classical Physics-Informed Neural Networks for Solving PDEs","date":"2025-03-20","arxiv_id":"2503.16678","repositories_listed":1,"syntology":null},{"url":"/paper/stop-overthinking-a-survey-on-efficient","slug":"stop-overthinking-a-survey-on-efficient","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","date":"2025-03-20","arxiv_id":"2503.16419","repositories_listed":1,"syntology":null},{"url":"/paper/the-emperor-s-new-clothes-in-benchmarking-a","slug":"the-emperor-s-new-clothes-in-benchmarking-a","title":"The Emperor's New Clothes in Benchmarking? A Rigorous Examination of Mitigation Strategies for LLM Benchmark Data Contamination","date":"2025-03-20","arxiv_id":"2503.16402","repositories_listed":1,"syntology":null},{"url":"/paper/language-based-image-colorization-a-benchmark","slug":"language-based-image-colorization-a-benchmark","title":"Language-based Image Colorization: A Benchmark and Beyond","date":"2025-03-19","arxiv_id":"2503.14974","repositories_listed":1,"syntology":null},{"url":"/paper/venusfactory-a-unified-platform-for-protein","slug":"venusfactory-a-unified-platform-for-protein","title":"VenusFactory: A Unified Platform for Protein Engineering Data Retrieval and Language Model Fine-Tuning","date":"2025-03-19","arxiv_id":"2503.15438","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-community-drug-response","slug":"benchmarking-community-drug-response","title":"Benchmarking community drug response prediction models: datasets, models, tools, and metrics for cross-dataset generalization analysis","date":"2025-03-18","arxiv_id":"2503.14356","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-failures-in-tool-augmented","slug":"benchmarking-failures-in-tool-augmented","title":"Benchmarking Failures in Tool-Augmented Language Models","date":"2025-03-18","arxiv_id":"2503.14227","repositories_listed":1,"syntology":null},{"url":"/paper/cospace-benchmarking-continuous-space","slug":"cospace-benchmarking-continuous-space","title":"CoSpace: Benchmarking Continuous Space Perception Ability for Vision-Language Models","date":"2025-03-18","arxiv_id":"2503.14161","repositories_listed":1,"syntology":null},{"url":"/paper/judge-benchmarking-judgment-document","slug":"judge-benchmarking-judgment-document","title":"JuDGE: Benchmarking Judgment Document Generation for Chinese Legal System","date":"2025-03-18","arxiv_id":"2503.14258","repositories_listed":1,"syntology":null},{"url":"/paper/microvqa-a-multimodal-reasoning-benchmark-for","slug":"microvqa-a-multimodal-reasoning-benchmark-for","title":"MicroVQA: A Multimodal Reasoning Benchmark for Microscopy-Based Scientific Research","date":"2025-03-17","arxiv_id":"2503.13399","repositories_listed":1,"syntology":null},{"url":"/paper/omnia-de-egotempo-benchmarking-temporal","slug":"omnia-de-egotempo-benchmarking-temporal","title":"Omnia de EgoTempo: Benchmarking Temporal Understanding of Multi-Modal LLMs in Egocentric Videos","date":"2025-03-17","arxiv_id":"2503.13646","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmarking-study-of-vision-based-robotic","slug":"a-benchmarking-study-of-vision-based-robotic","title":"A Benchmarking Study of Vision-based Robotic Grasping Algorithms","date":"2025-03-14","arxiv_id":"2503.11163","repositories_listed":1,"syntology":null},{"url":"/paper/gnns-as-predictors-of-agentic-workflow","slug":"gnns-as-predictors-of-agentic-workflow","title":"GNNs as Predictors of Agentic Workflow Performances","date":"2025-03-14","arxiv_id":"2503.11301","repositories_listed":1,"syntology":null},{"url":"/paper/vistai-benchmarking-vision-language-models","slug":"vistai-benchmarking-vision-language-models","title":"VisTai: Benchmarking Vision-Language Models for Traditional Chinese in Taiwan","date":"2025-03-13","arxiv_id":"2503.10427","repositories_listed":1,"syntology":null},{"url":"/paper/castle-benchmarking-dataset-for-static-code","slug":"castle-benchmarking-dataset-for-static-code","title":"CASTLE: Benchmarking Dataset for Static Code Analyzers and LLMs towards CWE Detection","date":"2025-03-12","arxiv_id":"2503.09433","repositories_listed":1,"syntology":null},{"url":"/paper/integration-of-nested-cross-validation","slug":"integration-of-nested-cross-validation","title":"Integration of nested cross-validation, automated hyperparameter optimization, high-performance computing to reduce and quantify the variance of test performance estimation of deep learning models","date":"2025-03-11","arxiv_id":"2503.08589","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-for-outpatient-referral","slug":"large-language-models-for-outpatient-referral","title":"Large Language Models for Outpatient Referral: Problem Definition, Benchmarking and Challenges","date":"2025-03-11","arxiv_id":"2503.08292","repositories_listed":1,"syntology":null},{"url":"/paper/robust-latent-matters-boosting-image","slug":"robust-latent-matters-boosting-image","title":"Robust Latent Matters: Boosting Image Generation with Sampling Error","date":"2025-03-11","arxiv_id":"2503.08354","repositories_listed":1,"syntology":null},{"url":"/paper/griffin-aerial-ground-cooperative-detection","slug":"griffin-aerial-ground-cooperative-detection","title":"Griffin: Aerial-Ground Cooperative Detection and Tracking Dataset and Benchmark","date":"2025-03-10","arxiv_id":"2503.06983","repositories_listed":1,"syntology":null},{"url":"/paper/illuminating-darkness-enhancing-real-world","slug":"illuminating-darkness-enhancing-real-world","title":"Illuminating Darkness: Enhancing Real-world Low-light Scenes with Smartphone Images","date":"2025-03-10","arxiv_id":"2503.06898","repositories_listed":1,"syntology":null},{"url":"/paper/medagentsbench-benchmarking-thinking-models","slug":"medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","arxiv_id":"2503.07459","repositories_listed":1,"syntology":{"n":23,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":4,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentsbench-benchmarking-thinking-models#ran","syntology_url":"https://syntology.ai/paper/2503.07459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07459"}},"official":{"repos":["gersteinlab/medagents-benchmark"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/skelite-compact-neural-networks-for-efficient","slug":"skelite-compact-neural-networks-for-efficient","title":"Skelite: Compact Neural Networks for Efficient Iterative Skeletonization","date":"2025-03-10","arxiv_id":"2503.07369","repositories_listed":1,"syntology":null},{"url":"/paper/dependeval-benchmarking-llms-for-repository","slug":"dependeval-benchmarking-llms-for-repository","title":"DependEval: Benchmarking LLMs for Repository Dependency Understanding","date":"2025-03-09","arxiv_id":"2503.06689","repositories_listed":1,"syntology":null},{"url":"/paper/dyncim-dynamic-curriculum-for-imbalanced","slug":"dyncim-dynamic-curriculum-for-imbalanced","title":"DynCIM: Dynamic Curriculum for Imbalanced Multimodal Learning","date":"2025-03-09","arxiv_id":"2503.06456","repositories_listed":1,"syntology":null},{"url":"/paper/knowlogic-a-benchmark-for-commonsense","slug":"knowlogic-a-benchmark-for-commonsense","title":"SCoRE: Benchmarking Long-Chain Reasoning in Commonsense Scenarios","date":"2025-03-08","arxiv_id":"2503.06218","repositories_listed":1,"syntology":null},{"url":"/paper/fedmabench-benchmarking-mobile-agents-on","slug":"fedmabench-benchmarking-mobile-agents-on","title":"FedMABench: Benchmarking Mobile Agents on Decentralized Heterogeneous User Data","date":"2025-03-07","arxiv_id":"2503.05143","repositories_listed":1,"syntology":null},{"url":"/paper/removing-geometric-bias-in-one-class-anomaly","slug":"removing-geometric-bias-in-one-class-anomaly","title":"Removing Geometric Bias in One-Class Anomaly Detection with Adaptive Feature Perturbation","date":"2025-03-07","arxiv_id":"2503.05520","repositories_listed":1,"syntology":null},{"url":"/paper/cldyb-towards-dynamic-benchmarking-for","slug":"cldyb-towards-dynamic-benchmarking-for","title":"CLDyB: Towards Dynamic Benchmarking for Continual Learning with Pre-trained Models","date":"2025-03-06","arxiv_id":"2503.04655","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-the-reasoning-abilities-of-llms","slug":"quantifying-the-reasoning-abilities-of-llms","title":"Quantifying the Reasoning Abilities of LLMs on Real-world Clinical Cases","date":"2025-03-06","arxiv_id":"2503.04691","repositories_listed":1,"syntology":null},{"url":"/paper/throwbench-benchmarking-llms-by-predicting","slug":"throwbench-benchmarking-llms-by-predicting","title":"ThrowBench: Benchmarking LLMs by Predicting Runtime Exceptions","date":"2025-03-06","arxiv_id":"2503.04241","repositories_listed":1,"syntology":null},{"url":"/paper/attackseqbench-benchmarking-large-language","slug":"attackseqbench-benchmarking-large-language","title":"AttackSeqBench: Benchmarking Large Language Models' Understanding of Sequential Patterns in Cyber Attacks","date":"2025-03-05","arxiv_id":"2503.03170","repositories_listed":1,"syntology":null},{"url":"/paper/gnnmerge-merging-of-gnn-models-without","slug":"gnnmerge-merging-of-gnn-models-without","title":"GNNMerge: Merging of GNN Models Without Accessing Training Data","date":"2025-03-05","arxiv_id":"2503.03384","repositories_listed":1,"syntology":null},{"url":"/paper/towards-universal-learning-based-model-for","slug":"towards-universal-learning-based-model-for","title":"Towards Universal Learning-based Model for Cardiac Image Reconstruction: Summary of the CMRxRecon2024 Challenge","date":"2025-03-05","arxiv_id":"2503.03971","repositories_listed":1,"syntology":null},{"url":"/paper/unpuzzle-a-unified-framework-for-pathology","slug":"unpuzzle-a-unified-framework-for-pathology","title":"UnPuzzle: A Unified Framework for Pathology Image Analysis","date":"2025-03-05","arxiv_id":"2503.03152","repositories_listed":1,"syntology":null},{"url":"/paper/2503-01306","slug":"2503-01306","title":"From Claims to Evidence: A Unified Framework and Critical Analysis of CNN vs. Transformer vs. Mamba in Medical Image Segmentation","date":"2025-03-03","arxiv_id":"2503.01306","repositories_listed":1,"syntology":null},{"url":"/paper/2503-01811","slug":"2503-01811","title":"AutoAdvExBench: Benchmarking autonomous exploitation of adversarial example defenses","date":"2025-03-03","arxiv_id":"2503.01811","repositories_listed":1,"syntology":null},{"url":"/paper/one-ruler-to-measure-them-all-benchmarking","slug":"one-ruler-to-measure-them-all-benchmarking","title":"One ruler to measure them all: Benchmarking multilingual long-context language models","date":"2025-03-03","arxiv_id":"2503.01996","repositories_listed":1,"syntology":null},{"url":"/paper/delving-into-out-of-distribution-detection-1","slug":"delving-into-out-of-distribution-detection-1","title":"Delving into Out-of-Distribution Detection with Medical Vision-Language Models","date":"2025-03-02","arxiv_id":"2503.01020","repositories_listed":1,"syntology":null},{"url":"/paper/2503-00089","slug":"2503-00089","title":"Protein Structure Tokenization: Benchmarking and New Recipe","date":"2025-02-28","arxiv_id":"2503.00089","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2503-00089#ran","syntology_url":"https://syntology.ai/paper/2503.00089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00089"}},"official":{"repos":["katarinayuan/structtokenbench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lexrag-benchmarking-retrieval-augmented","slug":"lexrag-benchmarking-retrieval-augmented","title":"LexRAG: Benchmarking Retrieval-Augmented Generation in Multi-Turn Legal Consultation Conversation","date":"2025-02-28","arxiv_id":"2502.20640","repositories_listed":1,"syntology":null},{"url":"/paper/neuromorse-a-temporally-structured-dataset","slug":"neuromorse-a-temporally-structured-dataset","title":"NeuroMorse: A Temporally Structured Dataset For Neuromorphic Computing","date":"2025-02-28","arxiv_id":"2502.20729","repositories_listed":1,"syntology":null},{"url":"/paper/collab-overcooked-benchmarking-and-evaluating","slug":"collab-overcooked-benchmarking-and-evaluating","title":"Collab-Overcooked: Benchmarking and Evaluating Large Language Models as Collaborative Agents","date":"2025-02-27","arxiv_id":"2502.20073","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collab-overcooked-benchmarking-and-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.20073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20073"}},"official":{"repos":["yusaemeow/collab-overcooked"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/egonormia-benchmarking-physical-social-norm","slug":"egonormia-benchmarking-physical-social-norm","title":"EgoNormia: Benchmarking Physical Social Norm Understanding","date":"2025-02-27","arxiv_id":"2502.20490","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egonormia-benchmarking-physical-social-norm#ran","syntology_url":"https://syntology.ai/paper/2502.20490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20490"}},"official":{"repos":["open-social-world/egonormia"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-learning-for-photoplethysmography","slug":"machine-learning-for-photoplethysmography","title":"Machine-learning for photoplethysmography analysis: Benchmarking feature, image, and signal-based approaches","date":"2025-02-27","arxiv_id":"2502.19949","repositories_listed":1,"syntology":null},{"url":"/paper/opentad-a-unified-framework-and-comprehensive","slug":"opentad-a-unified-framework-and-comprehensive","title":"OpenTAD: A Unified Framework and Comprehensive Study of Temporal Action Detection","date":"2025-02-27","arxiv_id":"2502.20361","repositories_listed":1,"syntology":null},{"url":"/paper/batterylife-a-comprehensive-dataset-and","slug":"batterylife-a-comprehensive-dataset-and","title":"BatteryLife: A Comprehensive Dataset and Benchmark for Battery Life Prediction","date":"2025-02-26","arxiv_id":"2502.18807","repositories_listed":1,"syntology":null},{"url":"/paper/codeif-benchmarking-the-instruction-following","slug":"codeif-benchmarking-the-instruction-following","title":"CodeIF: Benchmarking the Instruction-Following Capabilities of Large Language Models for Code Generation","date":"2025-02-26","arxiv_id":"2502.19166","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-graph-tasks-with-pure-llms-a","slug":"exploring-graph-tasks-with-pure-llms-a","title":"Exploring Graph Tasks with Pure LLMs: A Comprehensive Benchmark and Investigation","date":"2025-02-26","arxiv_id":"2502.18771","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/exploring-graph-tasks-with-pure-llms-a#ran","syntology_url":"https://syntology.ai/paper/2502.18771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18771"}},"official":{"repos":["myflashbarry/LLM-benchmarking"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/generalizable-deep-learning-for","slug":"generalizable-deep-learning-for","title":"Generalizable deep learning for photoplethysmography-based blood pressure estimation -- A Benchmarking Study","date":"2025-02-26","arxiv_id":"2502.19167","repositories_listed":1,"syntology":null},{"url":"/paper/medical-hallucinations-in-foundation-models","slug":"medical-hallucinations-in-foundation-models","title":"Medical Hallucinations in Foundation Models and Their Impact on Healthcare","date":"2025-02-26","arxiv_id":"2503.05777","repositories_listed":1,"syntology":null},{"url":"/paper/problem-solved-information-extraction-design","slug":"problem-solved-information-extraction-design","title":"Problem Solved? Information Extraction Design Space for Layout-Rich Documents using LLMs","date":"2025-02-25","arxiv_id":"2502.18179","repositories_listed":1,"syntology":null},{"url":"/paper/safe-multi-agent-navigation-guided-by-goal","slug":"safe-multi-agent-navigation-guided-by-goal","title":"Safe Multi-Agent Navigation guided by Goal-Conditioned Safe Reinforcement Learning","date":"2025-02-25","arxiv_id":"2502.17813","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-temporal-reasoning-and-alignment","slug":"benchmarking-temporal-reasoning-and-alignment","title":"Benchmarking Temporal Reasoning and Alignment Across Chinese Dynasties","date":"2025-02-24","arxiv_id":"2502.16922","repositories_listed":1,"syntology":null},{"url":"/paper/multitat-benchmarking-multilingual-table-and","slug":"multitat-benchmarking-multilingual-table-and","title":"MULTITAT: Benchmarking Multilingual Table-and-Text Question Answering","date":"2025-02-24","arxiv_id":"2502.17253","repositories_listed":1,"syntology":null},{"url":"/paper/an-analyst-inspector-framework-for-evaluating","slug":"an-analyst-inspector-framework-for-evaluating","title":"An Analyst-Inspector Framework for Evaluating Reproducibility of LLMs in Data Science","date":"2025-02-23","arxiv_id":"2502.16395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-analyst-inspector-framework-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.16395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16395"}},"official":{"repos":["qunhualilab/llm-ds-reproducibility"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/biomaze-benchmarking-and-enhancing-large","slug":"biomaze-benchmarking-and-enhancing-large","title":"BioMaze: Benchmarking and Enhancing Large Language Models for Biological Pathway Reasoning","date":"2025-02-23","arxiv_id":"2502.16660","repositories_listed":1,"syntology":null},{"url":"/paper/recent-advances-in-large-langauge-model","slug":"recent-advances-in-large-langauge-model","title":"Recent Advances in Large Langauge Model Benchmarks against Data Contamination: From Static to Dynamic Evaluation","date":"2025-02-23","arxiv_id":"2502.17521","repositories_listed":1,"syntology":null},{"url":"/paper/unmasking-societal-biases-in-respiratory","slug":"unmasking-societal-biases-in-respiratory","title":"Unmasking Societal Biases in Respiratory Support for ICU Patients through Social Determinants of Health","date":"2025-02-23","arxiv_id":"2502.16477","repositories_listed":1,"syntology":null},{"url":"/paper/visual-rag-benchmarking-text-to-image","slug":"visual-rag-benchmarking-text-to-image","title":"Visual-RAG: Benchmarking Text-to-Image Retrieval Augmented Generation for Visual Knowledge Intensive Queries","date":"2025-02-23","arxiv_id":"2502.16636","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-prompt-evaluation-systematic","slug":"adversarial-prompt-evaluation-systematic","title":"Adversarial Prompt Evaluation: Systematic Benchmarking of Guardrails Against Prompt Input Attacks on LLMs","date":"2025-02-21","arxiv_id":"2502.15427","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-prompt-evaluation-systematic#ran","syntology_url":"https://syntology.ai/paper/2502.15427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15427"}},"official":{"repos":["ibm/adversarial-prompt-evaluation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-machine-learning-for-bowel-sound","slug":"benchmarking-machine-learning-for-bowel-sound","title":"Benchmarking machine learning for bowel sound pattern classification from tabular features to pretrained models","date":"2025-02-21","arxiv_id":"2502.15607","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-rag-through-a-chart","slug":"benchmarking-multimodal-rag-through-a-chart","title":"Benchmarking Multimodal RAG through a Chart-based Document Question-Answering Generation Framework","date":"2025-02-20","arxiv_id":"2502.14864","repositories_listed":1,"syntology":null},{"url":"/paper/building-reliable-sim-driving-agents-by","slug":"building-reliable-sim-driving-agents-by","title":"Building reliable sim driving agents by scaling self-play","date":"2025-02-20","arxiv_id":"2502.14706","repositories_listed":1,"syntology":null},{"url":"/paper/fetalclip-a-visual-language-foundation-model","slug":"fetalclip-a-visual-language-foundation-model","title":"FetalCLIP: A Visual-Language Foundation Model for Fetal Ultrasound Image Analysis","date":"2025-02-20","arxiv_id":"2502.14807","repositories_listed":1,"syntology":null},{"url":"/paper/predictaboard-benchmarking-llm-score","slug":"predictaboard-benchmarking-llm-score","title":"PredictaBoard: Benchmarking LLM Score Predictability","date":"2025-02-20","arxiv_id":"2502.14445","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-porous-microstructures-automatic","slug":"synthetic-porous-microstructures-automatic","title":"Synthetic Porous Microstructures: Automatic Design, Simulation, and Permeability Analysis","date":"2025-02-20","arxiv_id":"2502.14518","repositories_listed":1,"syntology":null},{"url":"/paper/tritonbench-benchmarking-large-language-model","slug":"tritonbench-benchmarking-large-language-model","title":"TritonBench: Benchmarking Large Language Model Capabilities for Generating Triton Operators","date":"2025-02-20","arxiv_id":"2502.14752","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llms-for-political-science-a","slug":"benchmarking-llms-for-political-science-a","title":"Benchmarking LLMs for Political Science: A United Nations Perspective","date":"2025-02-19","arxiv_id":"2502.14122","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-self-supervised-methods-for","slug":"benchmarking-self-supervised-methods-for","title":"Benchmarking Self-Supervised Learning Methods for Accelerated MRI Reconstruction","date":"2025-02-19","arxiv_id":"2502.14009","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-learning-framework-for-efficient","slug":"a-deep-learning-framework-for-efficient","title":"A deep learning framework for efficient pathology image analysis","date":"2025-02-18","arxiv_id":"2502.13027","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-post-training-quantization-in","slug":"benchmarking-post-training-quantization-in","title":"Benchmarking Post-Training Quantization in LLMs: Comprehensive Taxonomy, Unified Evaluation, and Comparative Analysis","date":"2025-02-18","arxiv_id":"2502.13178","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-dynamic-resource-1","slug":"reinforcement-learning-for-dynamic-resource-1","title":"Reinforcement Learning for Dynamic Resource Allocation in Optical Networks: Hype or Hope?","date":"2025-02-18","arxiv_id":"2502.12804","repositories_listed":1,"syntology":null},{"url":"/paper/hintsoftruth-a-multimodal-checkworthiness","slug":"hintsoftruth-a-multimodal-checkworthiness","title":"HintsOfTruth: A Multimodal Checkworthiness Detection Dataset with Real and Synthetic Claims","date":"2025-02-17","arxiv_id":"2502.11753","repositories_listed":1,"syntology":null},{"url":"/paper/ilias-instance-level-image-retrieval-at-scale","slug":"ilias-instance-level-image-retrieval-at-scale","title":"ILIAS: Instance-Level Image retrieval At Scale","date":"2025-02-17","arxiv_id":"2502.11748","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ilias-instance-level-image-retrieval-at-scale#ran","syntology_url":"https://syntology.ai/paper/2502.11748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11748"}},"official":null}},{"url":"/paper/integrating-expert-knowledge-into-logical","slug":"integrating-expert-knowledge-into-logical","title":"Integrating Expert Knowledge into Logical Programs via LLMs","date":"2025-02-17","arxiv_id":"2502.12275","repositories_listed":1,"syntology":null},{"url":"/paper/positional-encoding-in-transformer-based-time","slug":"positional-encoding-in-transformer-based-time","title":"Positional Encoding in Transformer-Based Time Series Models: A Survey","date":"2025-02-17","arxiv_id":"2502.12370","repositories_listed":1,"syntology":null},{"url":"/paper/jexplore-design-space-exploration-tool-for","slug":"jexplore-design-space-exploration-tool-for","title":"JExplore: Design Space Exploration Tool for Nvidia Jetson Boards","date":"2025-02-16","arxiv_id":"2502.15773","repositories_listed":1,"syntology":null},{"url":"/paper/forecasting-time-series-with-constraints","slug":"forecasting-time-series-with-constraints","title":"Forecasting time series with constraints","date":"2025-02-14","arxiv_id":"2502.10485","repositories_listed":1,"syntology":null},{"url":"/paper/do-llms-recognize-your-preferences-evaluating","slug":"do-llms-recognize-your-preferences-evaluating","title":"Do LLMs Recognize Your Preferences? Evaluating Personalized Preference Following in LLMs","date":"2025-02-13","arxiv_id":"2502.09597","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/do-llms-recognize-your-preferences-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.09597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09597"}},"official":{"repos":["amazon-science/PrefEval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lob-bench-benchmarking-generative-ai-for","slug":"lob-bench-benchmarking-generative-ai-for","title":"LOB-Bench: Benchmarking Generative AI for Finance -- an Application to Limit Order Book Data","date":"2025-02-13","arxiv_id":"2502.09172","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-generation-of-synthetic","slug":"zero-shot-generation-of-synthetic","title":"Zero-shot generation of synthetic neurosurgical data with large language models","date":"2025-02-13","arxiv_id":"2502.09566","repositories_listed":1,"syntology":null},{"url":"/paper/fino1-on-the-transferability-of-reasoning","slug":"fino1-on-the-transferability-of-reasoning","title":"Fino1: On the Transferability of Reasoning Enhanced LLMs to Finance","date":"2025-02-12","arxiv_id":"2502.08127","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fino1-on-the-transferability-of-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.08127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.08127"}},"official":{"repos":["the-finai/fino1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exharmony-authorship-and-citations-for","slug":"exharmony-authorship-and-citations-for","title":"exHarmony: Authorship and Citations for Benchmarking the Reviewer Assignment Problem","date":"2025-02-11","arxiv_id":"2502.07683","repositories_listed":1,"syntology":null},{"url":"/paper/the-devil-is-in-the-prompts-de-identification","slug":"the-devil-is-in-the-prompts-de-identification","title":"The Devil is in the Prompts: De-Identification Traces Enhance Memorization Risks in Synthetic Chest X-Ray Generation","date":"2025-02-11","arxiv_id":"2502.07516","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-vision-language-models-on","slug":"benchmarking-vision-language-models-on","title":"Benchmarking Vision-Language Models on Optical Character Recognition in Dynamic Video Environments","date":"2025-02-10","arxiv_id":"2502.06445","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-vision-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2502.06445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06445"}},"official":{"repos":["video-db/ocr-benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-systematic-reasoning-abilities","slug":"evaluating-the-systematic-reasoning-abilities","title":"Evaluating the Systematic Reasoning Abilities of Large Language Models through Graph Coloring","date":"2025-02-10","arxiv_id":"2502.07087","repositories_listed":1,"syntology":null}],"record_sha256":"2b5f127082bd1c7af2531c6ef3cb500dc573114677ade45db320b297526a075d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}