{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/20","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":56,"rows_per_page":100,"rows":[1901,2000],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/19","next":"/task/benchmarking/papers/21","papers":[{"url":"/paper/mega-multilingual-evaluation-of-generative-ai","slug":"mega-multilingual-evaluation-of-generative-ai","title":"MEGA: Multilingual Evaluation of Generative AI","date":"2023-03-22","arxiv_id":"2303.12528","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mega-multilingual-evaluation-of-generative-ai#ran","syntology_url":"https://syntology.ai/paper/2303.12528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.12528"}},"official":null}},{"url":"/paper/deid-gpt-zero-shot-medical-text-de","slug":"deid-gpt-zero-shot-medical-text-de","title":"DeID-GPT: Zero-shot Medical Text De-Identification by GPT-4","date":"2023-03-20","arxiv_id":"2303.11032","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-realistic-test-time-training-1","slug":"revisiting-realistic-test-time-training-1","title":"Revisiting Realistic Test-Time Training: Sequential Inference and Adaptation by Anchored Clustering Regularized Self-Training","date":"2023-03-20","arxiv_id":"2303.10856","repositories_listed":1,"syntology":null},{"url":"/paper/cctv-gun-benchmarking-handgun-detection-in","slug":"cctv-gun-benchmarking-handgun-detection-in","title":"CCTV-Gun: Benchmarking Handgun Detection in CCTV Images","date":"2023-03-19","arxiv_id":"2303.10703","repositories_listed":1,"syntology":null},{"url":"/paper/from-mnist-to-imagenet-and-back-benchmarking","slug":"from-mnist-to-imagenet-and-back-benchmarking","title":"From MNIST to ImageNet and Back: Benchmarking Continual Curriculum Learning","date":"2023-03-16","arxiv_id":"2303.11076","repositories_listed":1,"syntology":null},{"url":"/paper/joint-multi-scale-tone-mapping-and-denoising","slug":"joint-multi-scale-tone-mapping-and-denoising","title":"Joint Multi-Scale Tone Mapping and Denoising for HDR Image Enhancement","date":"2023-03-16","arxiv_id":"2303.09071","repositories_listed":1,"syntology":null},{"url":"/paper/transnetr-transformer-based-residual-network","slug":"transnetr-transformer-based-residual-network","title":"TransNetR: Transformer-based Residual Network for Polyp Segmentation with Multi-Center Out-of-Distribution Testing","date":"2023-03-13","arxiv_id":"2303.07428","repositories_listed":1,"syntology":null},{"url":"/paper/aux-drop-handling-haphazard-inputs-in-online","slug":"aux-drop-handling-haphazard-inputs-in-online","title":"Aux-Drop: Handling Haphazard Inputs in Online Learning Using Auxiliary Dropouts","date":"2023-03-09","arxiv_id":"2303.05155","repositories_listed":1,"syntology":null},{"url":"/paper/badlad-a-large-multi-domain-bengali-document","slug":"badlad-a-large-multi-domain-bengali-document","title":"BaDLAD: A Large Multi-Domain Bengali Document Layout Analysis Dataset","date":"2023-03-09","arxiv_id":"2303.05325","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-multi-user-surface-recognition","slug":"multimodal-multi-user-surface-recognition","title":"Multimodal Multi-User Surface Recognition with the Kernel Two-Sample Test","date":"2023-03-08","arxiv_id":"2303.04930","repositories_listed":1,"syntology":null},{"url":"/paper/openoccupancy-a-large-scale-benchmark-for","slug":"openoccupancy-a-large-scale-benchmark-for","title":"OpenOccupancy: A Large Scale Benchmark for Surrounding Semantic Occupancy Perception","date":"2023-03-07","arxiv_id":"2303.03991","repositories_listed":1,"syntology":null},{"url":"/paper/extended-agriculture-vision-an-extension-of-a","slug":"extended-agriculture-vision-an-extension-of-a","title":"Extended Agriculture-Vision: An Extension of a Large Aerial Image Dataset for Agricultural Pattern Analysis","date":"2023-03-04","arxiv_id":"2303.02460","repositories_listed":1,"syntology":null},{"url":"/paper/fluidlab-a-differentiable-environment-for","slug":"fluidlab-a-differentiable-environment-for","title":"FluidLab: A Differentiable Environment for Benchmarking Complex Fluid Manipulation","date":"2023-03-04","arxiv_id":"2303.02346","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-framework-for-machine-learning","slug":"benchmarking-framework-for-machine-learning","title":"Benchmarking framework for machine learning classification from fNIRS data","date":"2023-03-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-white-blood-cell-classification","slug":"benchmarking-white-blood-cell-classification","title":"Benchmarking White Blood Cell Classification Under Domain Shift","date":"2023-03-03","arxiv_id":"2303.01777","repositories_listed":1,"syntology":null},{"url":"/paper/data-efficient-training-of-cnns-and","slug":"data-efficient-training-of-cnns-and","title":"Data-Efficient Training of CNNs and Transformers with Coresets: A Stability Perspective","date":"2023-03-03","arxiv_id":"2303.02095","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-self-supervised-contrastive","slug":"benchmarking-self-supervised-contrastive","title":"Benchmarking Self-Supervised Contrastive Learning Methods for Image-Based Plant Phenotyping","date":"2023-03-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sta-self-controlled-text-augmentation-for","slug":"sta-self-controlled-text-augmentation-for","title":"STA: Self-controlled Text Augmentation for Improving Text Classifications","date":"2023-02-24","arxiv_id":"2302.12784","repositories_listed":1,"syntology":null},{"url":"/paper/a-framework-for-benchmarking-class-out-of-1","slug":"a-framework-for-benchmarking-class-out-of-1","title":"A framework for benchmarking class-out-of-distribution detection and its application to ImageNet","date":"2023-02-23","arxiv_id":"2302.11893","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-framework-for-benchmarking-class-out-of-1#ran","syntology_url":"https://syntology.ai/paper/2302.11893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11893"}},"official":{"repos":["mdabbah/COOD_benchmarking"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dermatological-diagnosis-explainability","slug":"dermatological-diagnosis-explainability","title":"Dermatological Diagnosis Explainability Benchmark for Convolutional Neural Networks","date":"2023-02-23","arxiv_id":"2302.12084","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-the-gumbel-softmax-in-maddpg","slug":"revisiting-the-gumbel-softmax-in-maddpg","title":"Revisiting the Gumbel-Softmax in MADDPG","date":"2023-02-23","arxiv_id":"2302.11793","repositories_listed":1,"syntology":null},{"url":"/paper/what-can-we-learn-from-the-selective","slug":"what-can-we-learn-from-the-selective","title":"What Can We Learn From The Selective Prediction And Uncertainty Estimation Performance Of 523 Imagenet Classifiers","date":"2023-02-23","arxiv_id":"2302.11874","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/what-can-we-learn-from-the-selective#ran","syntology_url":"https://syntology.ai/paper/2302.11874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11874"}},"official":{"repos":["idogalil/benchmarking-uncertainty-estimation-performance"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/an-efficient-two-stage-gradient-boosting","slug":"an-efficient-two-stage-gradient-boosting","title":"An Efficient Two-stage Gradient Boosting Framework for Short-term Traffic State Estimation","date":"2023-02-21","arxiv_id":"2302.10400","repositories_listed":1,"syntology":null},{"url":"/paper/arena-rosnav-2-0-a-development-and","slug":"arena-rosnav-2-0-a-development-and","title":"Arena-Rosnav 2.0: A Development and Benchmarking Platform for Robot Navigation in Highly Dynamic Environments","date":"2023-02-20","arxiv_id":"2302.10023","repositories_listed":1,"syntology":null},{"url":"/paper/a-neuromorphic-dataset-for-object","slug":"a-neuromorphic-dataset-for-object","title":"A Neuromorphic Dataset for Object Segmentation in Indoor Cluttered Environment","date":"2023-02-13","arxiv_id":"2302.06301","repositories_listed":1,"syntology":null},{"url":"/paper/a-swat-based-reinforcement-learning-framework","slug":"a-swat-based-reinforcement-learning-framework","title":"A SWAT-based Reinforcement Learning Framework for Crop Management","date":"2023-02-10","arxiv_id":"2302.04988","repositories_listed":1,"syntology":null},{"url":"/paper/ai-sound-recognition-on-asthma-medication","slug":"ai-sound-recognition-on-asthma-medication","title":"AI Sound Recognition on Asthma Medication Adherence: Evaluation With the RDA Benchmark Suite","date":"2023-02-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fortuna-a-library-for-uncertainty","slug":"fortuna-a-library-for-uncertainty","title":"Fortuna: A Library for Uncertainty Quantification in Deep Learning","date":"2023-02-08","arxiv_id":"2302.04019","repositories_listed":1,"syntology":null},{"url":"/paper/surgt-soft-tissue-tracking-for-robotic","slug":"surgt-soft-tissue-tracking-for-robotic","title":"SurgT challenge: Benchmark of Soft-Tissue Trackers for Robotic Surgery","date":"2023-02-06","arxiv_id":"2302.03022","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-algorithms-for-submodular","slug":"benchmarking-algorithms-for-submodular","title":"Benchmarking Algorithms for Submodular Optimization Problems Using IOHProfiler","date":"2023-02-02","arxiv_id":"2302.01464","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-probabilistic-deep-learning","slug":"benchmarking-probabilistic-deep-learning","title":"Benchmarking Probabilistic Deep Learning Methods for License Plate Recognition","date":"2023-02-02","arxiv_id":"2302.01427","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-low-cost-microscopy-workflow-image","slug":"rethinking-low-cost-microscopy-workflow-image","title":"Rethinking low-cost microscopy workflow: Image enhancement using deep based Extended Depth of Field methods","date":"2023-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-for-news","slug":"benchmarking-large-language-models-for-news","title":"Benchmarking Large Language Models for News Summarization","date":"2023-01-31","arxiv_id":"2301.13848","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-hyper-to-real-space-projections","slug":"enhancing-hyper-to-real-space-projections","title":"Enhancing Hyper-To-Real Space Projections Through Euclidean Norm Meta-Heuristic Optimization","date":"2023-01-31","arxiv_id":"2301.13671","repositories_listed":1,"syntology":null},{"url":"/paper/population-wise-labeling-of-sulcal-graphs","slug":"population-wise-labeling-of-sulcal-graphs","title":"Population-wise Labeling of Sulcal Graphs using Multi-graph Matching","date":"2023-01-31","arxiv_id":"2301.13532","repositories_listed":1,"syntology":null},{"url":"/paper/sport-task-fine-grained-action-detection-and","slug":"sport-task-fine-grained-action-detection-and","title":"Sport Task: Fine Grained Action Detection and Classification of Table Tennis Strokes from Videos for MediaEval 2022","date":"2023-01-31","arxiv_id":"2301.13576","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-optimality-of-time-series","slug":"benchmarking-optimality-of-time-series","title":"Benchmarking optimality of time series classification methods in distinguishing diffusions","date":"2023-01-30","arxiv_id":"2301.13112","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-robustness-to-adversarial-image","slug":"benchmarking-robustness-to-adversarial-image","title":"Benchmarking Robustness to Adversarial Image Obfuscations","date":"2023-01-30","arxiv_id":"2301.12993","repositories_listed":1,"syntology":null},{"url":"/paper/quality-indicators-for-preference-based","slug":"quality-indicators-for-preference-based","title":"Quality Indicators for Preference-based Evolutionary Multi-objective Optimization Using a Reference Point: A Review and Analysis","date":"2023-01-28","arxiv_id":"2301.12148","repositories_listed":1,"syntology":null},{"url":"/paper/temporai-facilitating-machine-learning","slug":"temporai-facilitating-machine-learning","title":"TemporAI: Facilitating Machine Learning Innovation in Time Domain Tasks for Medicine","date":"2023-01-28","arxiv_id":"2301.12260","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporai-facilitating-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2301.12260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12260"}},"official":{"repos":["vanderschaarlab/temporai"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-graph-neural-network-evaluation","slug":"task-agnostic-graph-neural-network-evaluation","title":"Task-Agnostic Graph Neural Network Evaluation via Adversarial Collaboration","date":"2023-01-27","arxiv_id":"2301.11517","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/task-agnostic-graph-neural-network-evaluation#ran","syntology_url":"https://syntology.ai/paper/2301.11517","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.11517"}},"official":{"repos":["victorzxy/graphac"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-systematic-review-of-green-ai","slug":"a-systematic-review-of-green-ai","title":"A Systematic Review of Green AI","date":"2023-01-26","arxiv_id":"2301.11047","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-intrinsic-reward-shaping-for","slug":"automatic-intrinsic-reward-shaping-for","title":"Automatic Intrinsic Reward Shaping for Exploration in Deep Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.10886","repositories_listed":1,"syntology":null},{"url":"/paper/bibench-benchmarking-and-analyzing-network","slug":"bibench-benchmarking-and-analyzing-network","title":"BiBench: Benchmarking and Analyzing Network Binarization","date":"2023-01-26","arxiv_id":"2301.11233","repositories_listed":1,"syntology":null},{"url":"/paper/towards-robust-metrics-for-concept","slug":"towards-robust-metrics-for-concept","title":"Towards Robust Metrics for Concept Representation Evaluation","date":"2023-01-25","arxiv_id":"2301.10367","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-yolov5-and-yolov7-models-with","slug":"benchmarking-yolov5-and-yolov7-models-with","title":"Benchmarking YOLOv5 and YOLOv7 models with DeepSORT for droplet tracking applications","date":"2023-01-19","arxiv_id":"2301.08189","repositories_listed":1,"syntology":null},{"url":"/paper/desbordante-from-benchmarking-suite-to-high","slug":"desbordante-from-benchmarking-suite-to-high","title":"Desbordante: from benchmarking suite to high-performance science-intensive data profiler (preprint)","date":"2023-01-14","arxiv_id":"2301.05965","repositories_listed":1,"syntology":null},{"url":"/paper/young-labeled-faces-in-the-wild-ylfw-a","slug":"young-labeled-faces-in-the-wild-ylfw-a","title":"Young Labeled Faces in the Wild (YLFW): A Dataset for Children Faces Recognition","date":"2023-01-13","arxiv_id":"2301.05776","repositories_listed":1,"syntology":null},{"url":"/paper/critical-review-of-conformational-b-cell","slug":"critical-review-of-conformational-b-cell","title":"Critical review of conformational B-cell epitope prediction methods","date":"2023-01-10","arxiv_id":"2301.03878","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-transferability-of-machine","slug":"evaluating-the-transferability-of-machine","title":"Evaluating the Transferability of Machine-Learned Force Fields for Material Property Modeling","date":"2023-01-10","arxiv_id":"2301.03729","repositories_listed":1,"syntology":null},{"url":"/paper/anna-abstractive-text-to-image-synthesis-with","slug":"anna-abstractive-text-to-image-synthesis-with","title":"ANNA: Abstractive Text-to-Image Synthesis with Filtered News Captions","date":"2023-01-05","arxiv_id":"2301.02160","repositories_listed":1,"syntology":null},{"url":"/paper/trace-encoding-in-process-mining-a-survey-and","slug":"trace-encoding-in-process-mining-a-survey-and","title":"Trace Encoding in Process Mining: a survey and benchmarking","date":"2023-01-05","arxiv_id":"2301.02167","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-the-robustness-of-lidar-semantic","slug":"benchmarking-the-robustness-of-lidar-semantic","title":"Benchmarking the Robustness of LiDAR Semantic Segmentation Models","date":"2023-01-03","arxiv_id":"2301.00970","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-the-robustness-of-lidar-semantic#ran","syntology_url":"https://syntology.ai/paper/2301.00970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.00970"}},"official":{"repos":["yanx27/2dpass"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/effective-and-efficient-training-for-1","slug":"effective-and-efficient-training-for-1","title":"Improving Sequential Recommendation Models with an Enhanced Loss Function","date":"2023-01-03","arxiv_id":"2301.00979","repositories_listed":1,"syntology":null},{"url":"/paper/reference-twice-a-simple-and-unified-baseline","slug":"reference-twice-a-simple-and-unified-baseline","title":"Reference Twice: A Simple and Unified Baseline for Few-Shot Instance Segmentation","date":"2023-01-03","arxiv_id":"2301.01156","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-robustness-of-3d-object-1","slug":"benchmarking-robustness-of-3d-object-1","title":"Benchmarking Robustness of 3D Object Detection to Common Corruptions","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/migperf-a-comprehensive-benchmark-for-deep","slug":"migperf-a-comprehensive-benchmark-for-deep","title":"MIGPerf: A Comprehensive Benchmark for Deep Learning Training and Inference Workloads on Multi-Instance GPUs","date":"2023-01-01","arxiv_id":"2301.00407","repositories_listed":1,"syntology":null},{"url":"/paper/sqad-automatic-smartphone-camera-quality","slug":"sqad-automatic-smartphone-camera-quality","title":"SQAD: Automatic Smartphone Camera Quality Assessment and Benchmarking","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multispider-towards-benchmarking-multilingual","slug":"multispider-towards-benchmarking-multilingual","title":"MultiSpider: Towards Benchmarking Multilingual Text-to-SQL Semantic Parsing","date":"2022-12-27","arxiv_id":"2212.13492","repositories_listed":1,"syntology":null},{"url":"/paper/ultra-high-definition-low-light-image","slug":"ultra-high-definition-low-light-image","title":"Ultra-High-Definition Low-Light Image Enhancement: A Benchmark and Transformer-Based Method","date":"2022-12-22","arxiv_id":"2212.11548","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ultra-high-definition-low-light-image#ran","syntology_url":"https://syntology.ai/paper/2212.11548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.11548"}},"official":{"repos":["taowangzj/llformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/a-comprehensive-study-and-comparison-of-the","slug":"a-comprehensive-study-and-comparison-of-the","title":"A Comprehensive Study of the Robustness for LiDAR-based 3D Object Detectors against Adversarial Attacks","date":"2022-12-20","arxiv_id":"2212.10230","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-comprehensive-study-and-comparison-of-the#ran","syntology_url":"https://syntology.ai/paper/2212.10230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10230"}},"official":{"repos":["Eaphan/Robust3DOD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-person-re-identification-1","slug":"benchmarking-person-re-identification-1","title":"Benchmarking person re-identification datasets and approaches for practical real-world implementations","date":"2022-12-20","arxiv_id":"2212.09981","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-spatial-relationships-in-text-to","slug":"benchmarking-spatial-relationships-in-text-to","title":"Benchmarking Spatial Relationships in Text-to-Image Generation","date":"2022-12-20","arxiv_id":"2212.10015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-spatial-relationships-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2212.10015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10015"}},"official":{"repos":["microsoft/VISOR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-stereotypes-in-language-models","slug":"understanding-stereotypes-in-language-models","title":"Causally Testing Gender Bias in LLMs: A Case Study on Occupational Bias","date":"2022-12-20","arxiv_id":"2212.10678","repositories_listed":1,"syntology":null},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/benchmarking-large-language-models-for","slug":"benchmarking-large-language-models-for","title":"Benchmarking Large Language Models for Automated Verilog RTL Code Generation","date":"2022-12-13","arxiv_id":"2212.11140","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2212.11140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.11140"}},"official":{"repos":["shailja-thakur/vgen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-pre-training-for-visuo-motor-control","slug":"on-pre-training-for-visuo-motor-control","title":"On Pre-Training for Visuo-Motor Control: Revisiting a Learning-from-Scratch Baseline","date":"2022-12-12","arxiv_id":"2212.05749","repositories_listed":1,"syntology":null},{"url":"/paper/pypop7-a-pure-python-library-for-population","slug":"pypop7-a-pure-python-library-for-population","title":"PyPop7: A Pure-Python Library for Population-Based Black-Box Optimization","date":"2022-12-12","arxiv_id":"2212.05652","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-self-supervised-learning-on","slug":"benchmarking-self-supervised-learning-on","title":"Benchmarking Self-Supervised Learning on Diverse Pathology Datasets","date":"2022-12-09","arxiv_id":"2212.04690","repositories_listed":1,"syntology":null},{"url":"/paper/ego-body-pose-estimation-via-ego-head-pose","slug":"ego-body-pose-estimation-via-ego-head-pose","title":"Ego-Body Pose Estimation via Ego-Head Pose Estimation","date":"2022-12-09","arxiv_id":"2212.04636","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":11,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 2 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ego-body-pose-estimation-via-ego-head-pose#ran","syntology_url":"https://syntology.ai/paper/2212.04636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.04636"}},"official":null}},{"url":"/paper/an-open-unified-deep-graph-learning-framework","slug":"an-open-unified-deep-graph-learning-framework","title":"An open unified deep graph learning framework for discovering drug leads","date":"2022-12-06","arxiv_id":"2301.03424","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-automl-algorithms-on-a","slug":"benchmarking-automl-algorithms-on-a","title":"Benchmarking AutoML algorithms on a collection of synthetic classification problems","date":"2022-12-06","arxiv_id":"2212.02704","repositories_listed":1,"syntology":null},{"url":"/paper/dfee-interactive-dataflow-execution-and","slug":"dfee-interactive-dataflow-execution-and","title":"DFEE: Interactive DataFlow Execution and Evaluation Kit","date":"2022-12-04","arxiv_id":"2212.08099","repositories_listed":1,"syntology":null},{"url":"/paper/rlogist-fast-observation-strategy-on-whole","slug":"rlogist-fast-observation-strategy-on-whole","title":"RLogist: Fast Observation Strategy on Whole-slide Images with Deep Reinforcement Learning","date":"2022-12-04","arxiv_id":"2212.01737","repositories_listed":1,"syntology":null},{"url":"/paper/towards-scene-understanding-for-autonomous","slug":"towards-scene-understanding-for-autonomous","title":"Towards Scene Understanding for Autonomous Operations on Airport Aprons","date":"2022-12-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchenas-a-benchmarking-platform-for-1","slug":"benchenas-a-benchmarking-platform-for-1","title":"BenchENAS: A Benchmarking Platform for Evolutionary Neural Architecture Search","date":"2022-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/geoclidean-few-shot-generalization-in","slug":"geoclidean-few-shot-generalization-in","title":"Geoclidean: Few-Shot Generalization in Euclidean Geometry","date":"2022-11-30","arxiv_id":"2211.16663","repositories_listed":1,"syntology":null},{"url":"/paper/adsorbml-accelerating-adsorption-energy","slug":"adsorbml-accelerating-adsorption-energy","title":"AdsorbML: A Leap in Efficiency for Adsorption Energy Calculations using Generalizable Machine Learning Potentials","date":"2022-11-29","arxiv_id":"2211.16486","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adsorbml-accelerating-adsorption-energy#ran","syntology_url":"https://syntology.ai/paper/2211.16486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.16486"}},"official":{"repos":["open-catalyst-project/adsorbml"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/why-do-tree-based-models-still-outperform-1","slug":"why-do-tree-based-models-still-outperform-1","title":"Why do tree-based models still outperform deep learning on typical tabular data?","date":"2022-11-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/immersive-neural-graphics-primitives","slug":"immersive-neural-graphics-primitives","title":"Immersive Neural Graphics Primitives","date":"2022-11-24","arxiv_id":"2211.13494","repositories_listed":1,"syntology":null},{"url":"/paper/multi-mask-aggregators-for-graph-neural","slug":"multi-mask-aggregators-for-graph-neural","title":"Multi-Mask Aggregators for Graph Neural Networks","date":"2022-11-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/fseval-a-benchmarking-framework-for-feature","slug":"fseval-a-benchmarking-framework-for-feature","title":"fseval: A Benchmarking Framework for Feature Selection and Feature Ranking Algorithms","date":"2022-11-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/this-is-the-way-designing-and-compiling","slug":"this-is-the-way-designing-and-compiling","title":"This is the way: designing and compiling LEPISZCZE, a comprehensive NLP benchmark for Polish","date":"2022-11-23","arxiv_id":"2211.13112","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/this-is-the-way-designing-and-compiling#ran","syntology_url":"https://syntology.ai/paper/2211.13112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.13112"}},"official":{"repos":["clarin-pl/lepiszcze"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/l3cube-mahasbert-and-hindsbert-sentence-bert","slug":"l3cube-mahasbert-and-hindsbert-sentence-bert","title":"L3Cube-MahaSBERT and HindSBERT: Sentence BERT Models and Benchmarking BERT Sentence Representations for Hindi and Marathi","date":"2022-11-21","arxiv_id":"2211.11187","repositories_listed":1,"syntology":null},{"url":"/paper/cryptopt-verified-compilation-with-random","slug":"cryptopt-verified-compilation-with-random","title":"CryptOpt: Verified Compilation with Randomized Program Search for Cryptographic Primitives (full version)","date":"2022-11-19","arxiv_id":"2211.10665","repositories_listed":1,"syntology":null},{"url":"/paper/lidar-gait-benchmarking-3d-gait-recognition","slug":"lidar-gait-benchmarking-3d-gait-recognition","title":"LidarGait: Benchmarking 3D Gait Recognition with Point Clouds","date":"2022-11-19","arxiv_id":"2211.10598","repositories_listed":1,"syntology":null},{"url":"/paper/pic4rl-gym-a-ros2-modular-framework-for","slug":"pic4rl-gym-a-ros2-modular-framework-for","title":"PIC4rl-gym: a ROS2 modular framework for Robots Autonomous Navigation with Deep Reinforcement Learning","date":"2022-11-19","arxiv_id":"2211.10714","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-graph-neural-networks-for-fmri","slug":"benchmarking-graph-neural-networks-for-fmri","title":"Benchmarking Graph Neural Networks for FMRI analysis","date":"2022-11-16","arxiv_id":"2211.08927","repositories_listed":1,"syntology":null},{"url":"/paper/deep-emotion-recognition-in-textual","slug":"deep-emotion-recognition-in-textual","title":"Deep Emotion Recognition in Textual Conversations: A Survey","date":"2022-11-16","arxiv_id":"2211.09172","repositories_listed":1,"syntology":null},{"url":"/paper/harmonization-benchmarking-tool-for","slug":"harmonization-benchmarking-tool-for","title":"Harmonization Benchmarking Tool for Neuroimaging Datasets","date":"2022-11-15","arxiv_id":"2211.07869","repositories_listed":1,"syntology":null},{"url":"/paper/dealing-with-missing-data-using-attention-and","slug":"dealing-with-missing-data-using-attention-and","title":"Dealing with missing data using attention and latent space regularization","date":"2022-11-14","arxiv_id":"2211.07059","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmarking-dataset-with-2440-organic","slug":"a-benchmarking-dataset-with-2440-organic","title":"A Benchmarking Dataset with 2440 Organic Molecules for Volume Distribution at Steady State","date":"2022-11-10","arxiv_id":"2211.05661","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameter-optimization-in-deep-multi","slug":"hyperparameter-optimization-in-deep-multi","title":"Hyperparameter optimization in deep multi-target prediction","date":"2022-11-08","arxiv_id":"2211.04362","repositories_listed":1,"syntology":null},{"url":"/paper/okapi-generalising-better-by-making","slug":"okapi-generalising-better-by-making","title":"Okapi: Generalising Better by Making Statistical Matches Match","date":"2022-11-07","arxiv_id":"2211.05236","repositories_listed":1,"syntology":null},{"url":"/paper/improved-target-specific-stance-detection-on","slug":"improved-target-specific-stance-detection-on","title":"Improved Target-specific Stance Detection on Social Media Platforms by Delving into Conversation Threads","date":"2022-11-06","arxiv_id":"2211.03061","repositories_listed":1,"syntology":null},{"url":"/paper/eventea-benchmarking-entity-alignment-for","slug":"eventea-benchmarking-entity-alignment-for","title":"EventEA: Benchmarking Entity Alignment for Event-centric Knowledge Graphs","date":"2022-11-05","arxiv_id":"2211.02817","repositories_listed":1,"syntology":null},{"url":"/paper/the-legal-argument-reasoning-task-in-civil","slug":"the-legal-argument-reasoning-task-in-civil","title":"The Legal Argument Reasoning Task in Civil Procedure","date":"2022-11-05","arxiv_id":"2211.02950","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-legal-argument-reasoning-task-in-civil#ran","syntology_url":"https://syntology.ai/paper/2211.02950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.02950"}},"official":{"repos":["trusthlt/legal-argument-reasoning-task"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-quality-diversity-algorithms-on","slug":"benchmarking-quality-diversity-algorithms-on","title":"Benchmarking Quality-Diversity Algorithms on Neuroevolution for Reinforcement Learning","date":"2022-11-04","arxiv_id":"2211.02193","repositories_listed":1,"syntology":null},{"url":"/paper/signing-outside-the-studio-benchmarking","slug":"signing-outside-the-studio-benchmarking","title":"Signing Outside the Studio: Benchmarking Background Robustness for Continuous Sign Language Recognition","date":"2022-11-01","arxiv_id":"2211.00448","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-adversarial-patch-against-aerial","slug":"benchmarking-adversarial-patch-against-aerial","title":"Benchmarking Adversarial Patch Against Aerial Detection","date":"2022-10-30","arxiv_id":"2210.16765","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-adversarial-patch-against-aerial#ran","syntology_url":"https://syntology.ai/paper/2210.16765","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.16765"}},"official":{"repos":["jiaweilian/ap-pa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"4d2f261773c9f8586c4a9c5c7877f353041d6e955d3377066ebc435c9edbdbbd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}