{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/19","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":19,"pages_in_order":56,"rows_per_page":100,"rows":[1801,1900],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/18","next":"/task/benchmarking/papers/20","papers":[{"url":"/paper/labelbench-a-comprehensive-framework-for","slug":"labelbench-a-comprehensive-framework-for","title":"LabelBench: A Comprehensive Framework for Benchmarking Adaptive Label-Efficient Learning","date":"2023-06-16","arxiv_id":"2306.09910","repositories_listed":1,"syntology":null},{"url":"/paper/aqua-a-benchmarking-tool-for-label-quality-1","slug":"aqua-a-benchmarking-tool-for-label-quality-1","title":"AQuA: A Benchmarking Tool for Label Quality Assessment","date":"2023-06-15","arxiv_id":"2306.09467","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/aqua-a-benchmarking-tool-for-label-quality-1#ran","syntology_url":"https://syntology.ai/paper/2306.09467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09467"}},"official":{"repos":["autonlab/aqua"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/bed-bi-encoder-based-detectors-for-out-of","slug":"bed-bi-encoder-based-detectors-for-out-of","title":"BED: Bi-Encoder-Based Detectors for Out-of-Distribution Detection","date":"2023-06-15","arxiv_id":"2306.08852","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-of-using-real-world-sensory-inputs","slug":"challenges-of-using-real-world-sensory-inputs","title":"Towards Motion Forecasting with Real-World Perception Inputs: Are End-to-End Approaches Competitive?","date":"2023-06-15","arxiv_id":"2306.09281","repositories_listed":1,"syntology":null},{"url":"/paper/ffb-a-fair-fairness-benchmark-for-in","slug":"ffb-a-fair-fairness-benchmark-for-in","title":"FFB: A Fair Fairness Benchmark for In-Processing Group Fairness Methods","date":"2023-06-15","arxiv_id":"2306.09468","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ffb-a-fair-fairness-benchmark-for-in#ran","syntology_url":"https://syntology.ai/paper/2306.09468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09468"}},"official":{"repos":["ahxt/fair_fairness_benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/kola-carefully-benchmarking-world-knowledge","slug":"kola-carefully-benchmarking-world-knowledge","title":"KoLA: Carefully Benchmarking World Knowledge of Large Language Models","date":"2023-06-15","arxiv_id":"2306.09296","repositories_listed":1,"syntology":null},{"url":"/paper/mlonmcu-tinyml-benchmarking-with-fast","slug":"mlonmcu-tinyml-benchmarking-with-fast","title":"MLonMCU: TinyML Benchmarking with Fast Retargeting","date":"2023-06-15","arxiv_id":"2306.08951","repositories_listed":1,"syntology":null},{"url":"/paper/pareprop-fast-parallelized-reversible","slug":"pareprop-fast-parallelized-reversible","title":"PaReprop: Fast Parallelized Reversible Backpropagation","date":"2023-06-15","arxiv_id":"2306.09342","repositories_listed":1,"syntology":null},{"url":"/paper/re-benchmarking-pool-based-active-learning","slug":"re-benchmarking-pool-based-active-learning","title":"Re-Benchmarking Pool-Based Active Learning for Binary Classification","date":"2023-06-15","arxiv_id":"2306.08954","repositories_listed":1,"syntology":null},{"url":"/paper/symmetry-informed-geometric-representation-1","slug":"symmetry-informed-geometric-representation-1","title":"Symmetry-Informed Geometric Representation for Molecules, Proteins, and Crystalline Materials","date":"2023-06-15","arxiv_id":"2306.09375","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/symmetry-informed-geometric-representation-1#ran","syntology_url":"https://syntology.ai/paper/2306.09375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09375"}},"official":{"repos":["chao1224/geom3d"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-benchmarking-and-improving-the","slug":"towards-benchmarking-and-improving-the","title":"Towards Benchmarking and Improving the Temporal Reasoning Capability of Large Language Models","date":"2023-06-15","arxiv_id":"2306.08952","repositories_listed":1,"syntology":null},{"url":"/paper/detrex-benchmarking-detection-transformers","slug":"detrex-benchmarking-detection-transformers","title":"detrex: Benchmarking Detection Transformers","date":"2023-06-12","arxiv_id":"2306.07265","repositories_listed":1,"syntology":null},{"url":"/paper/aria-digital-twin-a-new-benchmark-dataset-for","slug":"aria-digital-twin-a-new-benchmark-dataset-for","title":"Aria Digital Twin: A New Benchmark Dataset for Egocentric 3D Machine Perception","date":"2023-06-10","arxiv_id":"2306.06362","repositories_listed":1,"syntology":null},{"url":"/paper/neurograph-benchmarks-for-graph-machine","slug":"neurograph-benchmarks-for-graph-machine","title":"NeuroGraph: Benchmarks for Graph Machine Learning in Brain Connectomics","date":"2023-06-09","arxiv_id":"2306.06202","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neurograph-benchmarks-for-graph-machine#ran","syntology_url":"https://syntology.ai/paper/2306.06202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06202"}},"official":{"repos":["anwar-said/neurograph"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dlama-a-framework-for-curating-culturally","slug":"dlama-a-framework-for-curating-culturally","title":"DLAMA: A Framework for Curating Culturally Diverse Facts for Probing the Knowledge of Pretrained Language Models","date":"2023-06-08","arxiv_id":"2306.05076","repositories_listed":1,"syntology":null},{"url":"/paper/dynamorep-trajectory-based-population","slug":"dynamorep-trajectory-based-population","title":"DynamoRep: Trajectory-Based Population Dynamics for Classification of Black-box Optimization Problems","date":"2023-06-08","arxiv_id":"2306.05438","repositories_listed":1,"syntology":null},{"url":"/paper/fedmlsecurity-a-benchmark-for-attacks-and","slug":"fedmlsecurity-a-benchmark-for-attacks-and","title":"FedSecurity: Benchmarking Attacks and Defenses in Federated Learning and Federated LLMs","date":"2023-06-08","arxiv_id":"2306.04959","repositories_listed":1,"syntology":null},{"url":"/paper/reference-matters-benchmarking-factual-error","slug":"reference-matters-benchmarking-factual-error","title":"Reference Matters: Benchmarking Factual Error Correction for Dialogue Summarization with Fine-grained Evaluation Framework","date":"2023-06-08","arxiv_id":"2306.05119","repositories_listed":1,"syntology":null},{"url":"/paper/knowing-how-knowing-that-a-new-task-for","slug":"knowing-how-knowing-that-a-new-task-for","title":"Knowing-how & Knowing-that: A New Task for Machine Comprehension of User Manuals","date":"2023-06-07","arxiv_id":"2306.04187","repositories_listed":1,"syntology":null},{"url":"/paper/self-adjusting-weighted-expected-improvement","slug":"self-adjusting-weighted-expected-improvement","title":"Self-Adjusting Weighted Expected Improvement for Bayesian Optimization","date":"2023-06-07","arxiv_id":"2306.04262","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-large-language-models-on-cmexam#ran","syntology_url":"https://syntology.ai/paper/2306.03030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03030"}},"official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-middle-trained-language-models","slug":"benchmarking-middle-trained-language-models","title":"Benchmarking Middle-Trained Language Models for Neural Search","date":"2023-06-05","arxiv_id":"2306.02867","repositories_listed":1,"syntology":null},{"url":"/paper/libauc-a-deep-learning-library-for-x-risk","slug":"libauc-a-deep-learning-library-for-x-risk","title":"LibAUC: A Deep Learning Library for X-Risk Optimization","date":"2023-06-05","arxiv_id":"2306.03065","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/libauc-a-deep-learning-library-for-x-risk#ran","syntology_url":"https://syntology.ai/paper/2306.03065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03065"}},"official":{"repos":["Optimization-AI/LibAUC"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/repobench-benchmarking-repository-level-code","slug":"repobench-benchmarking-repository-level-code","title":"RepoBench: Benchmarking Repository-Level Code Auto-Completion Systems","date":"2023-06-05","arxiv_id":"2306.03091","repositories_listed":1,"syntology":null},{"url":"/paper/score-based-enhanced-sampling-for-protein","slug":"score-based-enhanced-sampling-for-protein","title":"Str2Str: A Score-based Framework for Zero-shot Protein Conformation Sampling","date":"2023-06-05","arxiv_id":"2306.03117","repositories_listed":1,"syntology":{"n":24,"n_ran":22,"n_constructed":0,"n_ran_checked":20,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":3,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/score-based-enhanced-sampling-for-protein#ran","syntology_url":"https://syntology.ai/paper/2306.03117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03117"}},"official":{"repos":["lujiarui/str2str"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":20,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transdocanalyser-a-framework-for-offline-semi","slug":"transdocanalyser-a-framework-for-offline-semi","title":"TransDocAnalyser: A Framework for Offline Semi-structured Handwritten Document Analysis in the Legal Domain","date":"2023-06-03","arxiv_id":"2306.02142","repositories_listed":1,"syntology":null},{"url":"/paper/babyslm-language-acquisition-friendly","slug":"babyslm-language-acquisition-friendly","title":"BabySLM: language-acquisition-friendly benchmark of self-supervised spoken language models","date":"2023-06-02","arxiv_id":"2306.01506","repositories_listed":1,"syntology":null},{"url":"/paper/multilingual-conceptual-coverage-in-text-to","slug":"multilingual-conceptual-coverage-in-text-to","title":"Multilingual Conceptual Coverage in Text-to-Image Models","date":"2023-06-02","arxiv_id":"2306.01735","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multilingual-conceptual-coverage-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2306.01735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01735"}},"official":{"repos":["michaelsaxon/cococrola"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spatially-resolved-gene-expression-prediction","slug":"spatially-resolved-gene-expression-prediction","title":"Spatially Resolved Gene Expression Prediction from H&E Histology Images via Bi-modal Contrastive Learning","date":"2023-06-02","arxiv_id":"2306.01859","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatially-resolved-gene-expression-prediction#ran","syntology_url":"https://syntology.ai/paper/2306.01859","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01859"}},"official":{"repos":["bowang-lab/bleep"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-knowledge-retrieval-with-multi","slug":"end-to-end-knowledge-retrieval-with-multi","title":"End-to-end Knowledge Retrieval with Multi-modal Queries","date":"2023-06-01","arxiv_id":"2306.00424","repositories_listed":1,"syntology":null},{"url":"/paper/improving-and-benchmarking-offline","slug":"improving-and-benchmarking-offline","title":"Improving and Benchmarking Offline Reinforcement Learning Algorithms","date":"2023-06-01","arxiv_id":"2306.00972","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-hate-speech-benchmarks-from-data","slug":"revisiting-hate-speech-benchmarks-from-data","title":"Revisiting Hate Speech Benchmarks: From Data Curation to System Deployment","date":"2023-06-01","arxiv_id":"2306.01105","repositories_listed":1,"syntology":null},{"url":"/paper/speech-self-supervised-representation","slug":"speech-self-supervised-representation","title":"Speech Self-Supervised Representation Benchmarking: Are We Doing it Right?","date":"2023-06-01","arxiv_id":"2306.00452","repositories_listed":1,"syntology":null},{"url":"/paper/handling-large-discrete-action-spaces-via","slug":"handling-large-discrete-action-spaces-via","title":"Dynamic Neighborhood Construction for Structured Large Discrete Action Spaces","date":"2023-05-31","arxiv_id":"2305.19891","repositories_listed":1,"syntology":null},{"url":"/paper/design-and-implementation-of-intelligent","slug":"design-and-implementation-of-intelligent","title":"Design and implementation of intelligent packet filtering in IoT microcontroller-based devices","date":"2023-05-30","arxiv_id":"2305.19214","repositories_listed":1,"syntology":null},{"url":"/paper/idtoolkit-a-toolkit-for-benchmarking-and","slug":"idtoolkit-a-toolkit-for-benchmarking-and","title":"IDToolkit: A Toolkit for Benchmarking and Developing Inverse Design Algorithms in Nanophotonics","date":"2023-05-30","arxiv_id":"2305.18978","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-ridesharing-darp-instances-based","slug":"large-scale-ridesharing-darp-instances-based","title":"Large-scale Ridesharing DARP Instances Based on Real Travel Demand","date":"2023-05-30","arxiv_id":"2305.18859","repositories_listed":1,"syntology":null},{"url":"/paper/ringer-rapid-conformer-generation-for","slug":"ringer-rapid-conformer-generation-for","title":"Accurate and Efficient Structural Ensemble Generation of Macrocyclic Peptides using Internal Coordinate Diffusion","date":"2023-05-30","arxiv_id":"2305.19800","repositories_listed":1,"syntology":null},{"url":"/paper/scone-benchmarking-negation-reasoning-in","slug":"scone-benchmarking-negation-reasoning-in","title":"ScoNe: Benchmarking Negation Reasoning in Language Models With Fine-Tuning and In-Context Learning","date":"2023-05-30","arxiv_id":"2305.19426","repositories_listed":1,"syntology":null},{"url":"/paper/sheetcopilot-bringing-software-productivity","slug":"sheetcopilot-bringing-software-productivity","title":"SheetCopilot: Bringing Software Productivity to the Next Level through Large Language Models","date":"2023-05-30","arxiv_id":"2305.19308","repositories_listed":1,"syntology":null},{"url":"/paper/shufflemix-improving-representations-via","slug":"shufflemix-improving-representations-via","title":"ShuffleMix: Improving Representations via Channel-Wise Shuffle of Interpolated Hidden States","date":"2023-05-30","arxiv_id":"2305.18684","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-the-underlying-meaning-of-multimodal","slug":"decoding-the-underlying-meaning-of-multimodal","title":"Decoding the Underlying Meaning of Multimodal Hateful Memes","date":"2023-05-28","arxiv_id":"2305.17678","repositories_listed":1,"syntology":null},{"url":"/paper/indl-a-new-datasets-and-benchmark-for-in","slug":"indl-a-new-datasets-and-benchmark-for-in","title":"InDL: A New Dataset and Benchmark for In-Diagram Logic Interpretation based on Visual Illusion","date":"2023-05-28","arxiv_id":"2305.17716","repositories_listed":1,"syntology":null},{"url":"/paper/based-benchmarking-analysis-and-structural","slug":"based-benchmarking-analysis-and-structural","title":"BASED: Benchmarking, Analysis, and Structural Estimation of Deblurring","date":"2023-05-27","arxiv_id":"2305.17477","repositories_listed":1,"syntology":null},{"url":"/paper/the-brain-tumor-segmentation-brats-challenge-2","slug":"the-brain-tumor-segmentation-brats-challenge-2","title":"The Brain Tumor Segmentation (BraTS) Challenge 2023: Focus on Pediatrics (CBTN-CONNECT-DIPGR-ASNR-MICCAI BraTS-PEDs)","date":"2023-05-26","arxiv_id":"2305.17033","repositories_listed":1,"syntology":null},{"url":"/paper/zero-is-not-hero-yet-benchmarking-zero-shot","slug":"zero-is-not-hero-yet-benchmarking-zero-shot","title":"Zero is Not Hero Yet: Benchmarking Zero-Shot Performance of LLMs for Financial Tasks","date":"2023-05-26","arxiv_id":"2305.16633","repositories_listed":1,"syntology":null},{"url":"/paper/css-a-large-scale-cross-schema-chinese-text","slug":"css-a-large-scale-cross-schema-chinese-text","title":"CSS: A Large-scale Cross-schema Chinese Text-to-SQL Medical Dataset","date":"2023-05-25","arxiv_id":"2305.15891","repositories_listed":1,"syntology":null},{"url":"/paper/keyposs-plug-and-play-facial-landmark","slug":"keyposs-plug-and-play-facial-landmark","title":"KeyPosS: Plug-and-Play Facial Landmark Detection through GPS-Inspired True-Range Multilateration","date":"2023-05-25","arxiv_id":"2305.16437","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-uav-detection-in-complex","slug":"vision-based-uav-detection-in-complex","title":"Investigation of UAV Detection in Images with Complex Backgrounds and Rainy Artifacts","date":"2023-05-25","arxiv_id":"2305.16450","repositories_listed":1,"syntology":null},{"url":"/paper/gpt4graph-can-large-language-models","slug":"gpt4graph-can-large-language-models","title":"GPT4Graph: Can Large Language Models Understand Graph Structured Data ? An Empirical Evaluation and Benchmarking","date":"2023-05-24","arxiv_id":"2305.15066","repositories_listed":1,"syntology":null},{"url":"/paper/empowering-llm-based-machine-translation-with","slug":"empowering-llm-based-machine-translation-with","title":"Benchmarking Machine Translation with Cultural Awareness","date":"2023-05-23","arxiv_id":"2305.14328","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-large-language-models-for-classical","slug":"exploring-large-language-models-for-classical","title":"Exploring Large Language Models for Classical Philology","date":"2023-05-23","arxiv_id":"2305.13698","repositories_listed":1,"syntology":null},{"url":"/paper/property-guided-generative-modelling-for","slug":"property-guided-generative-modelling-for","title":"Robust Model-Based Optimization for Challenging Fitness Landscapes","date":"2023-05-23","arxiv_id":"2305.13650","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/property-guided-generative-modelling-for#ran","syntology_url":"https://syntology.ai/paper/2305.13650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13650"}},"official":{"repos":["sabagh1994/pgvae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-massively-multi-domain-multilingual","slug":"towards-massively-multi-domain-multilingual","title":"ReadMe++: Benchmarking Multilingual Language Models for Multi-Domain Readability Assessment","date":"2023-05-23","arxiv_id":"2305.14463","repositories_listed":1,"syntology":null},{"url":"/paper/when-the-music-stops-tip-of-the-tongue","slug":"when-the-music-stops-tip-of-the-tongue","title":"When the Music Stops: Tip-of-the-Tongue Retrieval for Music","date":"2023-05-23","arxiv_id":"2305.14072","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-on-extremely-weakly-supervised","slug":"a-benchmark-on-extremely-weakly-supervised","title":"A Benchmark on Extremely Weakly Supervised Text Classification: Reconcile Seed Matching and Prompting Approaches","date":"2023-05-22","arxiv_id":"2305.12749","repositories_listed":1,"syntology":null},{"url":"/paper/element-aware-summarization-with-large","slug":"element-aware-summarization-with-large","title":"Element-aware Summarization with Large Language Models: Expert-aligned Evaluation and Chain-of-Thought Method","date":"2023-05-22","arxiv_id":"2305.13412","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/element-aware-summarization-with-large#ran","syntology_url":"https://syntology.ai/paper/2305.13412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13412"}},"official":{"repos":["alsace08/sumcot"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/entred-benchmarking-relation-extraction-with","slug":"entred-benchmarking-relation-extraction-with","title":"How Fragile is Relation Extraction under Entity Replacements?","date":"2023-05-22","arxiv_id":"2305.13551","repositories_listed":1,"syntology":null},{"url":"/paper/towards-benchmarking-and-assessing-visual-1","slug":"towards-benchmarking-and-assessing-visual-1","title":"Towards Benchmarking and Assessing Visual Naturalness of Physical World Adversarial Attacks","date":"2023-05-22","arxiv_id":"2305.12863","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-benchmarking-and-assessing-visual-1#ran","syntology_url":"https://syntology.ai/paper/2305.12863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12863"}},"official":{"repos":["zhangsn-19/pan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-task-understanding-through","slug":"evaluating-task-understanding-through","title":"Separating form and meaning: Using self-consistency to quantify task understanding across multiple senses","date":"2023-05-19","arxiv_id":"2305.11662","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-task-understanding-through#ran","syntology_url":"https://syntology.ai/paper/2305.11662","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11662"}},"official":{"repos":["xeniaohmer/multisense_consistency"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/visualizing-linguistic-diversity-of-text","slug":"visualizing-linguistic-diversity-of-text","title":"Visualizing Linguistic Diversity of Text Datasets Synthesized by Large Language Models","date":"2023-05-19","arxiv_id":"2305.11364","repositories_listed":1,"syntology":null},{"url":"/paper/x-iqe-explainable-image-quality-evaluation","slug":"x-iqe-explainable-image-quality-evaluation","title":"X-IQE: eXplainable Image Quality Evaluation for Text-to-Image Generation with Visual Large Language Models","date":"2023-05-18","arxiv_id":"2305.10843","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-google-research","slug":"an-empirical-study-on-google-research","title":"An Empirical Study on Google Research Football Multi-agent Scenarios","date":"2023-05-16","arxiv_id":"2305.09458","repositories_listed":1,"syntology":null},{"url":"/paper/infometic-an-informative-metric-for-reference","slug":"infometic-an-informative-metric-for-reference","title":"InfoMetIC: An Informative Metric for Reference-free Image Caption Evaluation","date":"2023-05-10","arxiv_id":"2305.06002","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/infometic-an-informative-metric-for-reference#ran","syntology_url":"https://syntology.ai/paper/2305.06002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.06002"}},"official":{"repos":["hawlyq/infometic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-in-biomedical-natural","slug":"large-language-models-in-biomedical-natural","title":"Benchmarking large language models for biomedical natural language processing applications and recommendations","date":"2023-05-10","arxiv_id":"2305.16326","repositories_listed":1,"syntology":null},{"url":"/paper/dexart-benchmarking-generalizable-dexterous","slug":"dexart-benchmarking-generalizable-dexterous","title":"DexArt: Benchmarking Generalizable Dexterous Manipulation with Articulated Objects","date":"2023-05-09","arxiv_id":"2305.05706","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dexart-benchmarking-generalizable-dexterous#ran","syntology_url":"https://syntology.ai/paper/2305.05706","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.05706"}},"official":{"repos":["Kami-code/dexart-release"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/torchbench-benchmarking-pytorch-with-high-api","slug":"torchbench-benchmarking-pytorch-with-high-api","title":"TorchBench: Benchmarking PyTorch with High API Surface Coverage","date":"2023-04-27","arxiv_id":"2304.14226","repositories_listed":1,"syntology":null},{"url":"/paper/on-pitfalls-of-textit-remove-and-retrain-data","slug":"on-pitfalls-of-textit-remove-and-retrain-data","title":"On Pitfalls of $\\textit{RemOve-And-Retrain}$: Data Processing Inequality Perspective","date":"2023-04-26","arxiv_id":"2304.13836","repositories_listed":1,"syntology":null},{"url":"/paper/imuposer-full-body-pose-estimation-using-imus","slug":"imuposer-full-body-pose-estimation-using-imus","title":"IMUPoser: Full-Body Pose Estimation using IMUs in Phones, Watches, and Earbuds","date":"2023-04-25","arxiv_id":"2304.12518","repositories_listed":1,"syntology":null},{"url":"/paper/mixnerf-memory-efficient-nerf-with-feature","slug":"mixnerf-memory-efficient-nerf-with-feature","title":"MF-NeRF: Memory Efficient NeRF with Mixed-Feature Hash Table","date":"2023-04-25","arxiv_id":"2304.12587","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-chatgpt-4-on-acr-radiation","slug":"benchmarking-chatgpt-4-on-acr-radiation","title":"Benchmarking ChatGPT-4 on ACR Radiation Oncology In-Training (TXIT) Exam and Red Journal Gray Zone Cases: Potentials and Challenges for AI-Assisted Medical Education and Decision Making in Radiation Oncology","date":"2023-04-24","arxiv_id":"2304.11957","repositories_listed":1,"syntology":null},{"url":"/paper/indiscernible-object-counting-in-underwater","slug":"indiscernible-object-counting-in-underwater","title":"RGB-D Indiscernible Object Counting in Underwater Scenes","date":"2023-04-23","arxiv_id":"2304.11677","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-low-shot-robustness-to-natural","slug":"benchmarking-low-shot-robustness-to-natural","title":"Benchmarking Low-Shot Robustness to Natural Distribution Shifts","date":"2023-04-21","arxiv_id":"2304.11263","repositories_listed":1,"syntology":null},{"url":"/paper/scoda-domain-adaptive-shape-completion-for","slug":"scoda-domain-adaptive-shape-completion-for","title":"SCoDA: Domain Adaptive Shape Completion for Real Scans","date":"2023-04-20","arxiv_id":"2304.10179","repositories_listed":1,"syntology":null},{"url":"/paper/depth-functions-for-partial-orders-with-a","slug":"depth-functions-for-partial-orders-with-a","title":"Depth Functions for Partial Orders with a Descriptive Analysis of Machine Learning Algorithms","date":"2023-04-19","arxiv_id":"2304.09872","repositories_listed":1,"syntology":null},{"url":"/paper/graph-neural-network-based-anomaly-detection-1","slug":"graph-neural-network-based-anomaly-detection-1","title":"Graph Neural Network-Based Anomaly Detection for River Network Systems","date":"2023-04-19","arxiv_id":"2304.09367","repositories_listed":1,"syntology":null},{"url":"/paper/the-ebible-corpus-data-and-model-benchmarks","slug":"the-ebible-corpus-data-and-model-benchmarks","title":"The eBible Corpus: Data and Model Benchmarks for Bible Translation for Low-Resource Languages","date":"2023-04-19","arxiv_id":"2304.09919","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-ebible-corpus-data-and-model-benchmarks#ran","syntology_url":"https://syntology.ai/paper/2304.09919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09919"}},"official":{"repos":["biblenlp/ebible-experiments"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comparison-of-image-denoising-methods","slug":"a-comparison-of-image-denoising-methods","title":"A Comparison of Image Denoising Methods","date":"2023-04-18","arxiv_id":"2304.08990","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-actor-critic-deep-reinforcement","slug":"benchmarking-actor-critic-deep-reinforcement","title":"Benchmarking Actor-Critic Deep Reinforcement Learning Algorithms for Robotics Control with Action Constraints","date":"2023-04-18","arxiv_id":"2304.08743","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-actor-critic-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2304.08743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08743"}},"official":{"repos":["omron-sinicx/action-constrained-rl-benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/computational-performance-aware-benchmarking","slug":"computational-performance-aware-benchmarking","title":"Towards Computational Performance Engineering for Unsupervised Concept Drift Detection -- Complexities, Benchmarking, Performance Analysis","date":"2023-04-17","arxiv_id":"2304.08319","repositories_listed":1,"syntology":null},{"url":"/paper/certifiable-black-box-attack-ensuring","slug":"certifiable-black-box-attack-ensuring","title":"Certifiable Black-Box Attacks with Randomized Adversarial Examples: Breaking Defenses with Provable Confidence","date":"2023-04-10","arxiv_id":"2304.04343","repositories_listed":1,"syntology":null},{"url":"/paper/espnet-st-v2-multipurpose-spoken-language","slug":"espnet-st-v2-multipurpose-spoken-language","title":"ESPnet-ST-v2: Multipurpose Spoken Language Translation Toolkit","date":"2023-04-10","arxiv_id":"2304.04596","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/espnet-st-v2-multipurpose-spoken-language#ran","syntology_url":"https://syntology.ai/paper/2304.04596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.04596"}},"official":{"repos":["espnet/espnet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neurobench-advancing-neuromorphic-computing","slug":"neurobench-advancing-neuromorphic-computing","title":"NeuroBench: A Framework for Benchmarking Neuromorphic Computing Algorithms and Systems","date":"2023-04-10","arxiv_id":"2304.04640","repositories_listed":1,"syntology":null},{"url":"/paper/robopianist-a-benchmark-for-high-dimensional","slug":"robopianist-a-benchmark-for-high-dimensional","title":"RoboPianist: Dexterous Piano Playing with Deep Reinforcement Learning","date":"2023-04-09","arxiv_id":"2304.04150","repositories_listed":1,"syntology":null},{"url":"/paper/simbaml-connecting-mechanistic-models-and","slug":"simbaml-connecting-mechanistic-models-and","title":"SimbaML: Connecting Mechanistic Models and Machine Learning with Augmented Data","date":"2023-04-08","arxiv_id":"2304.04000","repositories_listed":1,"syntology":null},{"url":"/paper/probing-conceptual-understanding-of-large","slug":"probing-conceptual-understanding-of-large","title":"Probing Conceptual Understanding of Large Visual-Language Models","date":"2023-04-07","arxiv_id":"2304.03659","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/probing-conceptual-understanding-of-large#ran","syntology_url":"https://syntology.ai/paper/2304.03659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.03659"}},"official":{"repos":["Maddy12/UnderstandingVisualTextModels"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-robustness-to-text-guided","slug":"benchmarking-robustness-to-text-guided","title":"Benchmarking Robustness to Text-Guided Corruptions","date":"2023-04-06","arxiv_id":"2304.02963","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-statistical-representations-of","slug":"interpretable-statistical-representations-of","title":"Interpretable statistical representations of neural population dynamics and geometry","date":"2023-04-06","arxiv_id":"2304.03376","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/interpretable-statistical-representations-of#ran","syntology_url":"https://syntology.ai/paper/2304.03376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.03376"}},"official":{"repos":["Dynamics-of-Neural-Systems-Lab/MARBLE"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/ihcv-discovery-of-hidden-time-dependent","slug":"ihcv-discovery-of-hidden-time-dependent","title":"IHCV: Discovery of Hidden Time-Dependent Control Variables in Non-Linear Dynamical Systems","date":"2023-04-05","arxiv_id":"2304.02443","repositories_listed":1,"syntology":null},{"url":"/paper/logonet-a-fine-grained-network-for-instance","slug":"logonet-a-fine-grained-network-for-instance","title":"LogoNet: a fine-grained network for instance-level logo sketch retrieval","date":"2023-04-05","arxiv_id":"2304.02214","repositories_listed":1,"syntology":null},{"url":"/paper/mmvc-learned-multi-mode-video-compression","slug":"mmvc-learned-multi-mode-video-compression","title":"MMVC: Learned Multi-Mode Video Compression with Block-based Prediction Mode Selection and Density-Adaptive Entropy Coding","date":"2023-04-05","arxiv_id":"2304.02273","repositories_listed":1,"syntology":null},{"url":"/paper/the-saudi-privacy-policy-dataset","slug":"the-saudi-privacy-policy-dataset","title":"The Saudi Privacy Policy Dataset","date":"2023-04-05","arxiv_id":"2304.02757","repositories_listed":1,"syntology":null},{"url":"/paper/slperf-a-unified-framework-for-benchmarking","slug":"slperf-a-unified-framework-for-benchmarking","title":"SLPerf: a Unified Framework for Benchmarking Split Learning","date":"2023-04-04","arxiv_id":"2304.01502","repositories_listed":1,"syntology":null},{"url":"/paper/spam-t5-benchmarking-large-language-models","slug":"spam-t5-benchmarking-large-language-models","title":"Spam-T5: Benchmarking Large Language Models for Few-Shot Email Spam Detection","date":"2023-04-03","arxiv_id":"2304.01238","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-for-vision-tasks-a","slug":"vision-language-models-for-vision-tasks-a","title":"Vision-Language Models for Vision Tasks: A Survey","date":"2023-04-03","arxiv_id":"2304.00685","repositories_listed":1,"syntology":null},{"url":"/paper/enrich-multi-purpose-dataset-for-benchmarking","slug":"enrich-multi-purpose-dataset-for-benchmarking","title":"ENRICH: Multi-purposE dataset for beNchmaRking In Computer vision and pHotogrammetry","date":"2023-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-scale-invariant-sorting-criterion-to-find-a-1","slug":"a-scale-invariant-sorting-criterion-to-find-a-1","title":"A Scale-Invariant Sorting Criterion to Find a Causal Order in Additive Noise Models","date":"2023-03-31","arxiv_id":"2303.18211","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/a-scale-invariant-sorting-criterion-to-find-a-1#ran","syntology_url":"https://syntology.ai/paper/2303.18211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.18211"}},"official":null}},{"url":"/paper/lacvit-a-label-aware-contrastive-training","slug":"lacvit-a-label-aware-contrastive-training","title":"LaCViT: A Label-aware Contrastive Fine-tuning Framework for Vision Transformers","date":"2023-03-31","arxiv_id":"2303.18013","repositories_listed":1,"syntology":null},{"url":"/paper/what-makes-for-effective-few-shot-point-cloud","slug":"what-makes-for-effective-few-shot-point-cloud","title":"What Makes for Effective Few-shot Point Cloud Classification?","date":"2023-03-31","arxiv_id":"2304.00022","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-makes-for-effective-few-shot-point-cloud#ran","syntology_url":"https://syntology.ai/paper/2304.00022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.00022"}},"official":{"repos":["cgye96/a_closer_look_at_3dfsl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/balancing-policy-constraint-and-ensemble-size","slug":"balancing-policy-constraint-and-ensemble-size","title":"Balancing policy constraint and ensemble size in uncertainty-based offline reinforcement learning","date":"2023-03-26","arxiv_id":"2303.14716","repositories_listed":1,"syntology":null}],"record_sha256":"19c78fadea1312223e3a6a49ec454b6ff0125b60ac44df56921369be3ccc806c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}