{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/15","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":15,"pages_in_order":56,"rows_per_page":100,"rows":[1401,1500],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/14","next":"/task/benchmarking/papers/16","papers":[{"url":"/paper/a-comparative-analysis-of-word-level-metric","slug":"a-comparative-analysis-of-word-level-metric","title":"A Comparative Analysis of Word-Level Metric Differential Privacy: Benchmarking The Privacy-Utility Trade-off","date":"2024-04-04","arxiv_id":"2404.03324","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-chatgpt-on-algorithmic-reasoning","slug":"benchmarking-chatgpt-on-algorithmic-reasoning","title":"Benchmarking ChatGPT on Algorithmic Reasoning","date":"2024-04-04","arxiv_id":"2404.03441","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-chatgpt-on-algorithmic-reasoning#ran","syntology_url":"https://syntology.ai/paper/2404.03441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03441"}},"official":{"repos":["mcleish7/clrs4lm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-parameter-control-methods-in","slug":"benchmarking-parameter-control-methods-in","title":"Benchmarking Parameter Control Methods in Differential Evolution for Mixed-Integer Black-Box Optimization","date":"2024-04-04","arxiv_id":"2404.03303","repositories_listed":1,"syntology":null},{"url":"/paper/no-zero-shot-without-exponential-data","slug":"no-zero-shot-without-exponential-data","title":"No \"Zero-Shot\" Without Exponential Data: Pretraining Concept Frequency Determines Multimodal Model Performance","date":"2024-04-04","arxiv_id":"2404.04125","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/no-zero-shot-without-exponential-data#ran","syntology_url":"https://syntology.ai/paper/2404.04125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04125"}},"official":{"repos":["bethgelab/frequency_determines_performance"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/outlier-efficient-hopfield-layers-for-large","slug":"outlier-efficient-hopfield-layers-for-large","title":"Outlier-Efficient Hopfield Layers for Large Transformer-Based Models","date":"2024-04-04","arxiv_id":"2404.03828","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":5,"n_ran_checked":5,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/outlier-efficient-hopfield-layers-for-large#ran","syntology_url":"https://syntology.ai/paper/2404.03828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03828"}},"official":{"repos":["magics-lab/outeffhop"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/paris3d-reasoning-based-3d-part-segmentation","slug":"paris3d-reasoning-based-3d-part-segmentation","title":"PARIS3D: Reasoning-based 3D Part Segmentation Using Large Multimodal Model","date":"2024-04-04","arxiv_id":"2404.03836","repositories_listed":1,"syntology":null},{"url":"/paper/schroedinger-s-threshold-when-the-auc-doesn-t","slug":"schroedinger-s-threshold-when-the-auc-doesn-t","title":"Schroedinger's Threshold: When the AUC doesn't predict Accuracy","date":"2024-04-04","arxiv_id":"2404.03344","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-for-2","slug":"benchmarking-large-language-models-for-2","title":"Benchmarking Large Language Models for Persian: A Preliminary Study Focusing on ChatGPT","date":"2024-04-03","arxiv_id":"2404.02403","repositories_listed":1,"syntology":null},{"url":"/paper/advancing-llm-reasoning-generalists-with","slug":"advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","arxiv_id":"2404.02078","repositories_listed":1,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-llm-reasoning-generalists-with#ran","syntology_url":"https://syntology.ai/paper/2404.02078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02078"}},"official":{"repos":["openbmb/eurus"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/atom-level-optical-chemical-structure","slug":"atom-level-optical-chemical-structure","title":"Atom-Level Optical Chemical Structure Recognition with Limited Supervision","date":"2024-04-02","arxiv_id":"2404.01743","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/atom-level-optical-chemical-structure#ran","syntology_url":"https://syntology.ai/paper/2404.01743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01743"}},"official":{"repos":["molden/atomlenz"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/patch-psychometrics-assisted-benchmarking-of","slug":"patch-psychometrics-assisted-benchmarking-of","title":"PATCH! {P}sychometrics-{A}ssis{T}ed Ben{CH}marking of Large Language Models against Human Populations: A Case Study of Proficiency in 8th Grade Mathematics","date":"2024-04-02","arxiv_id":"2404.01799","repositories_listed":1,"syntology":null},{"url":"/paper/prego-online-mistake-detection-in-procedural","slug":"prego-online-mistake-detection-in-procedural","title":"PREGO: online mistake detection in PRocedural EGOcentric videos","date":"2024-04-02","arxiv_id":"2404.01933","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prego-online-mistake-detection-in-procedural#ran","syntology_url":"https://syntology.ai/paper/2404.01933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01933"}},"official":{"repos":["aleflabo/prego"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-counterfactual-image-generation","slug":"benchmarking-counterfactual-image-generation","title":"Benchmarking Counterfactual Image Generation","date":"2024-03-29","arxiv_id":"2403.20287","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-counterfactual-image-generation#ran","syntology_url":"https://syntology.ai/paper/2403.20287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20287"}},"official":{"repos":["gulnazaki/counterfactual-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-the-robustness-of-temporal","slug":"benchmarking-the-robustness-of-temporal","title":"Benchmarking the Robustness of Temporal Action Detection Models Against Temporal Corruptions","date":"2024-03-29","arxiv_id":"2403.20254","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-the-robustness-of-temporal#ran","syntology_url":"https://syntology.ai/paper/2403.20254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20254"}},"official":{"repos":["alvin-zeng/temporal-robustness-benchmark"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/indibias-a-benchmark-dataset-to-measure","slug":"indibias-a-benchmark-dataset-to-measure","title":"IndiBias: A Benchmark Dataset to Measure Social Biases in Language Models for Indian Context","date":"2024-03-29","arxiv_id":"2403.20147","repositories_listed":1,"syntology":null},{"url":"/paper/are-large-language-models-good-at-utility","slug":"are-large-language-models-good-at-utility","title":"Are Large Language Models Good at Utility Judgments?","date":"2024-03-28","arxiv_id":"2403.19216","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/are-large-language-models-good-at-utility#ran","syntology_url":"https://syntology.ai/paper/2403.19216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19216"}},"official":{"repos":["ict-bigdatalab/utility_judgments"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-implicit-neural-representation","slug":"benchmarking-implicit-neural-representation","title":"Benchmarking Implicit Neural Representation and Geometric Rendering in Real-Time RGB-D SLAM","date":"2024-03-28","arxiv_id":"2403.19473","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-object-detectors-with-coco-a-new","slug":"benchmarking-object-detectors-with-coco-a-new","title":"Benchmarking Object Detectors with COCO: A New Path Forward","date":"2024-03-27","arxiv_id":"2403.18819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-object-detectors-with-coco-a-new#ran","syntology_url":"https://syntology.ai/paper/2403.18819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18819"}},"official":{"repos":["kdexd/coco-rem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imagenet-d-benchmarking-neural-network","slug":"imagenet-d-benchmarking-neural-network","title":"ImageNet-D: Benchmarking Neural Network Robustness on Diffusion Synthetic Object","date":"2024-03-27","arxiv_id":"2403.18775","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imagenet-d-benchmarking-neural-network#ran","syntology_url":"https://syntology.ai/paper/2403.18775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18775"}},"official":{"repos":["chenshuang-zhang/imagenet_d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rankmamba-benchmarking-mamba-s-document","slug":"rankmamba-benchmarking-mamba-s-document","title":"RankMamba: Benchmarking Mamba's Document Ranking Performance in the Era of Transformers","date":"2024-03-27","arxiv_id":"2403.18276","repositories_listed":1,"syntology":null},{"url":"/paper/towards-image-ambient-lighting-normalization","slug":"towards-image-ambient-lighting-normalization","title":"Towards Image Ambient Lighting Normalization","date":"2024-03-27","arxiv_id":"2403.18730","repositories_listed":1,"syntology":null},{"url":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic","slug":"arabicaqa-a-comprehensive-dataset-for-arabic","title":"ArabicaQA: A Comprehensive Dataset for Arabic Question Answering","date":"2024-03-26","arxiv_id":"2403.17848","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic#ran","syntology_url":"https://syntology.ai/paper/2403.17848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17848"}},"official":{"repos":["datascienceuibk/arabicaqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/addressing-the-generalization-of-3d","slug":"addressing-the-generalization-of-3d","title":"Addressing the generalization of 3D registration methods with a featureless baseline and an unbiased benchmark","date":"2024-03-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-the-fragility-of-active-learners","slug":"on-the-fragility-of-active-learners","title":"On the Fragility of Active Learners for Text Classification","date":"2024-03-23","arxiv_id":"2403.15744","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/on-the-fragility-of-active-learners#ran","syntology_url":"https://syntology.ai/paper/2403.15744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15744"}},"official":{"repos":["ThuongTNguyen/ALchemist"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-chinese-commonsense-reasoning-of","slug":"benchmarking-chinese-commonsense-reasoning-of","title":"Benchmarking Chinese Commonsense Reasoning of LLMs: From Chinese-Specifics to Reasoning-Memorization Correlations","date":"2024-03-21","arxiv_id":"2403.14112","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-chinese-commonsense-reasoning-of#ran","syntology_url":"https://syntology.ai/paper/2403.14112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14112"}},"official":{"repos":["opendatalab/charm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/can-3d-vision-language-models-truly","slug":"can-3d-vision-language-models-truly","title":"Can 3D Vision-Language Models Truly Understand Natural Language?","date":"2024-03-21","arxiv_id":"2403.14760","repositories_listed":1,"syntology":null},{"url":"/paper/domainlab-a-modular-python-package-for-domain","slug":"domainlab-a-modular-python-package-for-domain","title":"DomainLab: A modular Python package for domain generalization in deep learning","date":"2024-03-21","arxiv_id":"2403.14356","repositories_listed":1,"syntology":null},{"url":"/paper/rodla-benchmarking-the-robustness-of-document","slug":"rodla-benchmarking-the-robustness-of-document","title":"RoDLA: Benchmarking the Robustness of Document Layout Analysis Models","date":"2024-03-21","arxiv_id":"2403.14442","repositories_listed":1,"syntology":null},{"url":"/paper/practical-end-to-end-optical-music","slug":"practical-end-to-end-optical-music","title":"Practical End-to-End Optical Music Recognition for Pianoform Music","date":"2024-03-20","arxiv_id":"2403.13763","repositories_listed":1,"syntology":null},{"url":"/paper/alphafin-benchmarking-financial-analysis-with","slug":"alphafin-benchmarking-financial-analysis-with","title":"AlphaFin: Benchmarking Financial Analysis with Retrieval-Augmented Stock-Chain Framework","date":"2024-03-19","arxiv_id":"2403.12582","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alphafin-benchmarking-financial-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2403.12582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12582"}},"official":{"repos":["alphafin-proj/alphafin"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/marta-a-model-for-the-automatic-phonemic","slug":"marta-a-model-for-the-automatic-phonemic","title":"MARTA: a model for the automatic phonemic grouping of the parkinsonian speech","date":"2024-03-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/melting-point-mobile-evaluation-of-language","slug":"melting-point-mobile-evaluation-of-language","title":"MELTing point: Mobile Evaluation of Language Transformers","date":"2024-03-19","arxiv_id":"2403.12844","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/melting-point-mobile-evaluation-of-language#ran","syntology_url":"https://syntology.ai/paper/2403.12844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12844"}},"official":{"repos":["brave-experiments/melt-public"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-the-robustness-of-uav-tracking","slug":"benchmarking-the-robustness-of-uav-tracking","title":"Benchmarking the Robustness of UAV Tracking Against Common Corruptions","date":"2024-03-18","arxiv_id":"2403.11424","repositories_listed":1,"syntology":null},{"url":"/paper/novelqa-a-benchmark-for-long-range-novel","slug":"novelqa-a-benchmark-for-long-range-novel","title":"NovelQA: Benchmarking Question Answering on Documents Exceeding 200K Tokens","date":"2024-03-18","arxiv_id":"2403.12766","repositories_listed":1,"syntology":null},{"url":"/paper/an-improved-metric-and-benchmark-for","slug":"an-improved-metric-and-benchmark-for","title":"An Improved Metric and Benchmark for Assessing the Performance of Virtual Screening Models","date":"2024-03-15","arxiv_id":"2403.10478","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-improved-metric-and-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2403.10478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.10478"}},"official":{"repos":["molecularmodelinglab/bigbind"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-zero-shot-robustness-of","slug":"benchmarking-zero-shot-robustness-of","title":"Benchmarking Zero-Shot Robustness of Multimodal Foundation Models: A Pilot Study","date":"2024-03-15","arxiv_id":"2403.10499","repositories_listed":1,"syntology":null},{"url":"/paper/histo-genomic-knowledge-distillation-for","slug":"histo-genomic-knowledge-distillation-for","title":"Histo-Genomic Knowledge Distillation For Cancer Prognosis From Histopathology Whole Slide Images","date":"2024-03-15","arxiv_id":"2403.10040","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-class-conditioned-alignment","slug":"attention-based-class-conditioned-alignment","title":"Attention-based Class-Conditioned Alignment for Multi-Source Domain Adaptation of Object Detectors","date":"2024-03-14","arxiv_id":"2403.09918","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-drafter-for-fast-speculative","slug":"recurrent-drafter-for-fast-speculative","title":"Recurrent Drafter for Fast Speculative Decoding in Large Language Models","date":"2024-03-14","arxiv_id":"2403.09919","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-drafter-for-fast-speculative#ran","syntology_url":"https://syntology.ai/paper/2403.09919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09919"}},"official":{"repos":["apple/ml-recurrent-drafter"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spoken-100-a-cross-lingual-benchmarking","slug":"spoken-100-a-cross-lingual-benchmarking","title":"SpokeN-100: A Cross-Lingual Benchmarking Dataset for The Classification of Spoken Numbers in Different Languages","date":"2024-03-14","arxiv_id":"2403.09753","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-learning-for-anomaly-traffic","slug":"semi-supervised-learning-for-anomaly-traffic","title":"Semi-Supervised Learning for Anomaly Traffic Detection via Bidirectional Normalizing Flows","date":"2024-03-13","arxiv_id":"2403.10550","repositories_listed":1,"syntology":null},{"url":"/paper/amharic-llama-and-llava-multimodal-llms-for","slug":"amharic-llama-and-llava-multimodal-llms-for","title":"Amharic LLaMA and LLaVA: Multimodal LLMs for Low Resource Languages","date":"2024-03-11","arxiv_id":"2403.06354","repositories_listed":1,"syntology":null},{"url":"/paper/class-imbalance-in-object-detection-an","slug":"class-imbalance-in-object-detection-an","title":"Class Imbalance in Object Detection: An Experimental Diagnosis and Study of Mitigation Strategies","date":"2024-03-11","arxiv_id":"2403.07113","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-foundation-models-for-content","slug":"leveraging-foundation-models-for-content","title":"Leveraging Foundation Models for Content-Based Medical Image Retrieval in Radiology","date":"2024-03-11","arxiv_id":"2403.06567","repositories_listed":1,"syntology":null},{"url":"/paper/addressing-shortcomings-in-fair-graph","slug":"addressing-shortcomings-in-fair-graph","title":"Addressing Shortcomings in Fair Graph Learning Datasets: Towards a New Benchmark","date":"2024-03-09","arxiv_id":"2403.06017","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-hpc-framework-with-multi-gpu-enabled","slug":"quantum-hpc-framework-with-multi-gpu-enabled","title":"Multi-GPU-Enabled Hybrid Quantum-Classical Workflow in Quantum-HPC Middleware: Applications in Quantum Simulations","date":"2024-03-09","arxiv_id":"2403.05828","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-for-1","slug":"benchmarking-large-language-models-for-1","title":"Benchmarking Large Language Models for Molecule Prediction Tasks","date":"2024-03-08","arxiv_id":"2403.05075","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-micro-action-recognition-dataset","slug":"benchmarking-micro-action-recognition-dataset","title":"Benchmarking Micro-action Recognition: Dataset, Methods, and Applications","date":"2024-03-08","arxiv_id":"2403.05234","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-micro-action-recognition-dataset#ran","syntology_url":"https://syntology.ai/paper/2403.05234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05234"}},"official":{"repos":["vut-hfut/micro-action"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/synth4bench-a-framework-for-generating","slug":"synth4bench-a-framework-for-generating","title":"Synth4bench: a framework for generating synthetic genomics data for the evaluation of tumor-only somatic variant calling algorithms","date":"2024-03-08","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tapilot-crossing-benchmarking-and-evolving","slug":"tapilot-crossing-benchmarking-and-evolving","title":"Tapilot-Crossing: Benchmarking and Evolving LLMs Towards Interactive Data Analysis Agents","date":"2024-03-08","arxiv_id":"2403.05307","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tapilot-crossing-benchmarking-and-evolving#ran","syntology_url":"https://syntology.ai/paper/2403.05307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05307"}},"official":{"repos":["tapilot-crossing/tapilot_code"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dissecting-sample-hardness-a-fine-grained","slug":"dissecting-sample-hardness-a-fine-grained","title":"Dissecting Sample Hardness: A Fine-Grained Analysis of Hardness Characterization Methods for Data-Centric AI","date":"2024-03-07","arxiv_id":"2403.04551","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dissecting-sample-hardness-a-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2403.04551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04551"}},"official":{"repos":["seedatnabeel/h-cat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ducho-2-0-towards-a-more-up-to-date-feature","slug":"ducho-2-0-towards-a-more-up-to-date-feature","title":"Ducho 2.0: Towards a More Up-to-Date Unified Framework for the Extraction of Multimodal Features in Recommendation","date":"2024-03-07","arxiv_id":"2403.04503","repositories_listed":1,"syntology":null},{"url":"/paper/improvements-evaluations-on-the-mlcommons","slug":"improvements-evaluations-on-the-mlcommons","title":"Improvements & Evaluations on the MLCommons CloudMask Benchmark","date":"2024-03-07","arxiv_id":"2403.04553","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-hallucination-in-large-language","slug":"benchmarking-hallucination-in-large-language","title":"Benchmarking Hallucination in Large Language Models based on Unanswerable Math Word Problem","date":"2024-03-06","arxiv_id":"2403.03558","repositories_listed":1,"syntology":null},{"url":"/paper/comparison-performance-of-spectrogram-and","slug":"comparison-performance-of-spectrogram-and","title":"Comparison Performance of Spectrogram and Scalogram as Input of Acoustic Recognition Task","date":"2024-03-06","arxiv_id":"2403.03611","repositories_listed":1,"syntology":null},{"url":"/paper/three-revisits-to-node-level-graph-anomaly","slug":"three-revisits-to-node-level-graph-anomaly","title":"Three Revisits to Node-Level Graph Anomaly Detection: Outliers, Message Passing and Hyperbolic Neural Networks","date":"2024-03-06","arxiv_id":"2403.04010","repositories_listed":1,"syntology":null},{"url":"/paper/sciassess-benchmarking-llm-proficiency-in","slug":"sciassess-benchmarking-llm-proficiency-in","title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","date":"2024-03-04","arxiv_id":"2403.01976","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sciassess-benchmarking-llm-proficiency-in#ran","syntology_url":"https://syntology.ai/paper/2403.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01976"}},"official":{"repos":["sci-assess/sciassess"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-dcf-an-architecture-agnostic-metric-with","slug":"a-dcf-an-architecture-agnostic-metric-with","title":"a-DCF: an architecture agnostic metric with application to spoofing-robust speaker verification","date":"2024-03-03","arxiv_id":"2403.01355","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-segmentation-models-with-mask","slug":"benchmarking-segmentation-models-with-mask","title":"Benchmarking Segmentation Models with Mask-Preserved Attribute Editing","date":"2024-03-02","arxiv_id":"2403.01231","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-datasets-a-toolkit-for","slug":"imitation-learning-datasets-a-toolkit-for","title":"Imitation Learning Datasets: A Toolkit For Creating Datasets, Training Agents and Benchmarking","date":"2024-03-01","arxiv_id":"2403.00550","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-datasets-a-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/2403.00550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00550"}},"official":{"repos":["nathangavenski/il-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/private-benchmarking-to-prevent-contamination","slug":"private-benchmarking-to-prevent-contamination","title":"TRUCE: Private Benchmarking to Prevent Contamination and Improve Comparative Evaluation of LLMs","date":"2024-03-01","arxiv_id":"2403.00393","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/private-benchmarking-to-prevent-contamination#ran","syntology_url":"https://syntology.ai/paper/2403.00393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00393"}},"official":{"repos":["microsoft/private-benchmarking"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lifelong-benchmarks-efficient-model","slug":"lifelong-benchmarks-efficient-model","title":"Efficient Lifelong Model Evaluation in an Era of Rapid Progress","date":"2024-02-29","arxiv_id":"2402.19472","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lifelong-benchmarks-efficient-model#ran","syntology_url":"https://syntology.ai/paper/2402.19472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19472"}},"official":{"repos":["bethgelab/sort-and-search"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-2","slug":"benchmarking-large-language-models-on-2","title":"Benchmarking Large Language Models on Answering and Explaining Challenging Medical Questions","date":"2024-02-28","arxiv_id":"2402.18060","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-large-language-models-on-2#ran","syntology_url":"https://syntology.ai/paper/2402.18060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18060"}},"official":{"repos":["hanjiechen/challengeclinicalqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/editing-factual-knowledge-and-explanatory","slug":"editing-factual-knowledge-and-explanatory","title":"Editing Factual Knowledge and Explanatory Ability of Medical Large Language Models","date":"2024-02-28","arxiv_id":"2402.18099","repositories_listed":1,"syntology":null},{"url":"/paper/flowcyt-a-comparative-study-of-deep-learning","slug":"flowcyt-a-comparative-study-of-deep-learning","title":"FlowCyt: A Comparative Study of Deep Learning Approaches for Multi-Class Classification in Flow Cytometry Benchmarking","date":"2024-02-28","arxiv_id":"2403.00024","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/flowcyt-a-comparative-study-of-deep-learning#ran","syntology_url":"https://syntology.ai/paper/2403.00024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00024"}},"official":{"repos":["LorenzoBini4/FlowCyt-Classification-Benchmark"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/are-llms-capable-of-data-based-statistical","slug":"are-llms-capable-of-data-based-statistical","title":"Are LLMs Capable of Data-based Statistical and Causal Reasoning? Benchmarking Advanced Quantitative Reasoning with Data","date":"2024-02-27","arxiv_id":"2402.17644","repositories_listed":1,"syntology":null},{"url":"/paper/beacon-a-lightweight-deep-reinforcement","slug":"beacon-a-lightweight-deep-reinforcement","title":"Beacon, a lightweight deep reinforcement learning benchmark library for flow control","date":"2024-02-27","arxiv_id":"2402.17402","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-data-science-agents","slug":"benchmarking-data-science-agents","title":"Benchmarking Data Science Agents","date":"2024-02-27","arxiv_id":"2402.17168","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-data-science-agents#ran","syntology_url":"https://syntology.ai/paper/2402.17168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17168"}},"official":{"repos":["metacopilot/dseval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-kandy-benchmark-incremental-neuro","slug":"the-kandy-benchmark-incremental-neuro","title":"The KANDY Benchmark: Incremental Neuro-Symbolic Learning and Reasoning with Kandinsky Patterns","date":"2024-02-27","arxiv_id":"2402.17431","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-kandy-benchmark-incremental-neuro#ran","syntology_url":"https://syntology.ai/paper/2402.17431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17431"}},"official":{"repos":["continual-nesy/kandybenchmark"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/partial-rankings-of-optimizers","slug":"partial-rankings-of-optimizers","title":"Partial Rankings of Optimizers","date":"2024-02-26","arxiv_id":"2402.16565","repositories_listed":1,"syntology":null},{"url":"/paper/hypotermqa-hypothetical-terms-dataset-for","slug":"hypotermqa-hypothetical-terms-dataset-for","title":"HypoTermQA: Hypothetical Terms Dataset for Benchmarking Hallucination Tendency of LLMs","date":"2024-02-25","arxiv_id":"2402.16211","repositories_listed":1,"syntology":null},{"url":"/paper/pst-bench-tracing-and-benchmarking-the-source","slug":"pst-bench-tracing-and-benchmarking-the-source","title":"PST-Bench: Tracing and Benchmarking the Source of Publications","date":"2024-02-25","arxiv_id":"2402.16009","repositories_listed":1,"syntology":null},{"url":"/paper/api-blend-a-comprehensive-corpora-for","slug":"api-blend-a-comprehensive-corpora-for","title":"API-BLEND: A Comprehensive Corpora for Training and Benchmarking API LLMs","date":"2024-02-23","arxiv_id":"2402.15491","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/api-blend-a-comprehensive-corpora-for#ran","syntology_url":"https://syntology.ai/paper/2402.15491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15491"}},"official":{"repos":["ibm/api-blend"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tombench-benchmarking-theory-of-mind-in-large","slug":"tombench-benchmarking-theory-of-mind-in-large","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","date":"2024-02-23","arxiv_id":"2402.15052","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tombench-benchmarking-theory-of-mind-in-large#ran","syntology_url":"https://syntology.ai/paper/2402.15052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15052"}},"official":{"repos":["zhchen18/tombench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/criticbench-benchmarking-llms-for-critique","slug":"criticbench-benchmarking-llms-for-critique","title":"CriticBench: Benchmarking LLMs for Critique-Correct Reasoning","date":"2024-02-22","arxiv_id":"2402.14809","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/criticbench-benchmarking-llms-for-critique#ran","syntology_url":"https://syntology.ai/paper/2402.14809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14809"}},"official":{"repos":["CriticBench/CriticBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genception-evaluate-multimodal-llms-with","slug":"genception-evaluate-multimodal-llms-with","title":"GenCeption: Evaluate Multimodal LLMs with Unlabeled Unimodal Data","date":"2024-02-22","arxiv_id":"2402.14973","repositories_listed":1,"syntology":null},{"url":"/paper/is-llm-as-a-judge-robust-investigating","slug":"is-llm-as-a-judge-robust-investigating","title":"Is LLM-as-a-Judge Robust? Investigating Universal Adversarial Attacks on Zero-shot LLM Assessment","date":"2024-02-21","arxiv_id":"2402.14016","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-llm-as-a-judge-robust-investigating#ran","syntology_url":"https://syntology.ai/paper/2402.14016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14016"}},"official":{"repos":["rainavyas/attack-comparative-assessment"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-soc-benchmarking-multimodal-large-language","slug":"mm-soc-benchmarking-multimodal-large-language","title":"MM-Soc: Benchmarking Multimodal Large Language Models in Social Media Platforms","date":"2024-02-21","arxiv_id":"2402.14154","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-soc-benchmarking-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.14154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14154"}},"official":{"repos":["claws-lab/mmsoc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pqa-zero-shot-protein-question-answering-for","slug":"pqa-zero-shot-protein-question-answering-for","title":"PQA: Zero-shot Protein Question Answering for Free-form Scientific Enquiry with Large Language Models","date":"2024-02-21","arxiv_id":"2402.13653","repositories_listed":1,"syntology":null},{"url":"/paper/the-effect-of-batch-size-on-contrastive-self","slug":"the-effect-of-batch-size-on-contrastive-self","title":"The Effect of Batch Size on Contrastive Self-Supervised Speech Representation Learning","date":"2024-02-21","arxiv_id":"2402.13723","repositories_listed":1,"syntology":null},{"url":"/paper/chili-chemically-informed-large-scale","slug":"chili-chemically-informed-large-scale","title":"CHILI: Chemically-Informed Large-scale Inorganic Nanomaterials Dataset for Advancing Graph Machine Learning","date":"2024-02-20","arxiv_id":"2402.13221","repositories_listed":1,"syntology":null},{"url":"/paper/class-incremental-learning-for-time-series","slug":"class-incremental-learning-for-time-series","title":"Class-incremental Learning for Time Series: Benchmark and Evaluation","date":"2024-02-19","arxiv_id":"2402.12035","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/class-incremental-learning-for-time-series#ran","syntology_url":"https://syntology.ai/paper/2402.12035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12035"}},"official":{"repos":["zqiao11/tscil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/event-based-motion-magnification","slug":"event-based-motion-magnification","title":"Event-Based Motion Magnification","date":"2024-02-19","arxiv_id":"2402.11957","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-location-trajectory-generation","slug":"synthetic-location-trajectory-generation","title":"Synthetic location trajectory generation using categorical diffusion models","date":"2024-02-19","arxiv_id":"2402.12242","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-knowledge-boundary-for-large","slug":"benchmarking-knowledge-boundary-for-large","title":"Benchmarking Knowledge Boundary for Large Language Models: A Different Perspective on Model Evaluation","date":"2024-02-18","arxiv_id":"2402.11493","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-knowledge-boundary-for-large#ran","syntology_url":"https://syntology.ai/paper/2402.11493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11493"}},"official":{"repos":["pkulcwmzx/knowledge-boundary"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-zeroth-order-optimization-for","slug":"revisiting-zeroth-order-optimization-for","title":"Revisiting Zeroth-Order Optimization for Memory-Efficient LLM Fine-Tuning: A Benchmark","date":"2024-02-18","arxiv_id":"2402.11592","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":3,"n_instrument":6,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/revisiting-zeroth-order-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2402.11592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11592"}},"official":{"repos":["zo-bench/zo-llm"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/panda-pedantic-answer-correctness","slug":"panda-pedantic-answer-correctness","title":"PEDANTS: Cheap but Effective and Interpretable Answer Equivalence","date":"2024-02-17","arxiv_id":"2402.11161","repositories_listed":1,"syntology":null},{"url":"/paper/ai-hospital-interactive-evaluation-and","slug":"ai-hospital-interactive-evaluation-and","title":"AI Hospital: Benchmarking Large Language Models in a Multi-agent Medical Interaction Simulator","date":"2024-02-15","arxiv_id":"2402.09742","repositories_listed":1,"syntology":null},{"url":"/paper/from-variability-to-stability-advancing","slug":"from-variability-to-stability-advancing","title":"From Variability to Stability: Advancing RecSys Benchmarking Practices","date":"2024-02-15","arxiv_id":"2402.09766","repositories_listed":1,"syntology":null},{"url":"/paper/sawec-sensing-assisted-wireless-edge","slug":"sawec-sensing-assisted-wireless-edge","title":"SAWEC: Sensing-Assisted Wireless Edge Computing","date":"2024-02-15","arxiv_id":"2402.10021","repositories_listed":1,"syntology":null},{"url":"/paper/the-butterfly-effect-of-model-editing-few","slug":"the-butterfly-effect-of-model-editing-few","title":"The Butterfly Effect of Model Editing: Few Edits Can Trigger Large Language Models Collapse","date":"2024-02-15","arxiv_id":"2402.09656","repositories_listed":1,"syntology":null},{"url":"/paper/massively-multi-cultural-knowledge","slug":"massively-multi-cultural-knowledge","title":"Massively Multi-Cultural Knowledge Acquisition & LM Benchmarking","date":"2024-02-14","arxiv_id":"2402.09369","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/massively-multi-cultural-knowledge#ran","syntology_url":"https://syntology.ai/paper/2402.09369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09369"}},"official":{"repos":["yrf1/llm-massivemulticulturenormsknowledge-nclb"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/multimedeval-a-benchmark-and-a-toolkit-for","slug":"multimedeval-a-benchmark-and-a-toolkit-for","title":"MultiMedEval: A Benchmark and a Toolkit for Evaluating Medical Vision-Language Models","date":"2024-02-14","arxiv_id":"2402.09262","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimedeval-a-benchmark-and-a-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/2402.09262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09262"}},"official":{"repos":["corentin-ryr/multimedeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bdslw60-a-word-level-bangla-sign-language","slug":"bdslw60-a-word-level-bangla-sign-language","title":"BdSLW60: A Word-Level Bangla Sign Language Dataset","date":"2024-02-13","arxiv_id":"2402.08635","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multi-component-signal","slug":"benchmarking-multi-component-signal","title":"Benchmarking multi-component signal processing methods in the time-frequency plane","date":"2024-02-13","arxiv_id":"2402.08521","repositories_listed":1,"syntology":null},{"url":"/paper/lota-bench-benchmarking-language-oriented","slug":"lota-bench-benchmarking-language-oriented","title":"LoTa-Bench: Benchmarking Language-oriented Task Planners for Embodied Agents","date":"2024-02-13","arxiv_id":"2402.08178","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lota-bench-benchmarking-language-oriented#ran","syntology_url":"https://syntology.ai/paper/2402.08178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08178"}},"official":{"repos":["lbaa2022/llmtaskplanning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/air-bench-benchmarking-large-audio-language","slug":"air-bench-benchmarking-large-audio-language","title":"AIR-Bench: Benchmarking Large Audio-Language Models via Generative Comprehension","date":"2024-02-12","arxiv_id":"2402.07729","repositories_listed":1,"syntology":null},{"url":"/paper/customizable-perturbation-synthesis-for","slug":"customizable-perturbation-synthesis-for","title":"Customizable Perturbation Synthesis for Robust SLAM Benchmarking","date":"2024-02-12","arxiv_id":"2402.08125","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/customizable-perturbation-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2402.08125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08125"}},"official":{"repos":["xiaohao-xu/slam-under-perturbation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-tree-based-approaches-surpass-deep","slug":"can-tree-based-approaches-surpass-deep","title":"Can Tree Based Approaches Surpass Deep Learning in Anomaly Detection? A Benchmarking Study","date":"2024-02-11","arxiv_id":"2402.07281","repositories_listed":1,"syntology":null}],"record_sha256":"cc60b0dd1a1815f1d9aa3cadd8bce42ae8c1620f15add9c54a6e9e1652c5fffe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}