{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/18","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":56,"rows_per_page":100,"rows":[1701,1800],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/17","next":"/task/benchmarking/papers/19","papers":[{"url":"/paper/benchmarking-encoder-decoder-architectures","slug":"benchmarking-encoder-decoder-architectures","title":"Benchmarking Encoder-Decoder Architectures for Biplanar X-ray to 3D Shape Reconstruction","date":"2023-09-24","arxiv_id":"2309.13587","repositories_listed":1,"syntology":null},{"url":"/paper/machine-assisted-mixed-methods-augmenting","slug":"machine-assisted-mixed-methods-augmenting","title":"Machine-assisted quantitizing designs: augmenting humanities and social sciences with artificial intelligence","date":"2023-09-24","arxiv_id":"2309.14379","repositories_listed":1,"syntology":null},{"url":"/paper/grad-dft-a-software-library-for-machine","slug":"grad-dft-a-software-library-for-machine","title":"Grad DFT: a software library for machine learning enhanced density functional theory","date":"2023-09-23","arxiv_id":"2309.15127","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-thematic-investment-with-prompt","slug":"accelerating-thematic-investment-with-prompt","title":"Prompt Tuned Embedding Classification for Multi-Label Industry Sector Allocation","date":"2023-09-21","arxiv_id":"2309.12075","repositories_listed":1,"syntology":null},{"url":"/paper/early-diagnosis-of-autism-spectrum-disorder","slug":"early-diagnosis-of-autism-spectrum-disorder","title":"An Evaluation of Machine Learning Approaches for Early Diagnosis of Autism Spectrum Disorder","date":"2023-09-20","arxiv_id":"2309.11646","repositories_listed":1,"syntology":null},{"url":"/paper/anchor-points-benchmarking-models-with-much","slug":"anchor-points-benchmarking-models-with-much","title":"Anchor Points: Benchmarking Models with Much Fewer Examples","date":"2023-09-14","arxiv_id":"2309.08638","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/anchor-points-benchmarking-models-with-much#ran","syntology_url":"https://syntology.ai/paper/2309.08638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.08638"}},"official":{"repos":["rvivek3/anchorpoints"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/m3dsynth-a-dataset-of-medical-3d-images-with","slug":"m3dsynth-a-dataset-of-medical-3d-images-with","title":"M3Dsynth: A dataset of medical 3D images with AI-generated local manipulations","date":"2023-09-14","arxiv_id":"2309.07973","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m3dsynth-a-dataset-of-medical-3d-images-with#ran","syntology_url":"https://syntology.ai/paper/2309.07973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07973"}},"official":null}},{"url":"/paper/verilogeval-evaluating-large-language-models","slug":"verilogeval-evaluating-large-language-models","title":"VerilogEval: Evaluating Large Language Models for Verilog Code Generation","date":"2023-09-14","arxiv_id":"2309.07544","repositories_listed":1,"syntology":null},{"url":"/paper/an-image-dataset-for-benchmarking-recommender","slug":"an-image-dataset-for-benchmarking-recommender","title":"An Image Dataset for Benchmarking Recommender Systems with Raw Pixels","date":"2023-09-13","arxiv_id":"2309.06789","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-image-dataset-for-benchmarking-recommender#ran","syntology_url":"https://syntology.ai/paper/2309.06789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06789"}},"official":{"repos":["westlake-repl/pixelrec"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-procedural-language","slug":"benchmarking-procedural-language","title":"Benchmarking Procedural Language Understanding for Low-Resource Languages: A Case Study on Turkish","date":"2023-09-13","arxiv_id":"2309.06698","repositories_listed":1,"syntology":null},{"url":"/paper/formalizing-multimedia-recommendation-through","slug":"formalizing-multimedia-recommendation-through","title":"Formalizing Multimedia Recommendation through Multimodal Deep Learning","date":"2023-09-11","arxiv_id":"2309.05273","repositories_listed":1,"syntology":null},{"url":"/paper/freeman-towards-benchmarking-3d-human-pose","slug":"freeman-towards-benchmarking-3d-human-pose","title":"FreeMan: Towards Benchmarking 3D Human Pose Estimation under Real-World Conditions","date":"2023-09-10","arxiv_id":"2309.05073","repositories_listed":1,"syntology":null},{"url":"/paper/recad-towards-a-unified-library-for","slug":"recad-towards-a-unified-library-for","title":"RecAD: Towards A Unified Library for Recommender Attack and Defense","date":"2023-09-09","arxiv_id":"2309.04884","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-out-of-distribution-electricity","slug":"navigating-out-of-distribution-electricity","title":"Navigating Out-of-Distribution Electricity Load Forecasting during COVID-19: Benchmarking energy load forecasting models without and with continual learning","date":"2023-09-08","arxiv_id":"2309.04296","repositories_listed":1,"syntology":null},{"url":"/paper/2309-03685","slug":"2309-03685","title":"PyGraft: Configurable Generation of Synthetic Schemas and Knowledge Graphs at Your Fingertips","date":"2023-09-07","arxiv_id":"2309.03685","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-of-large-language-models-for","slug":"evaluation-of-large-language-models-for","title":"Evaluation of large language models for discovery of gene set function","date":"2023-09-07","arxiv_id":"2309.04019","repositories_listed":1,"syntology":null},{"url":"/paper/learning-continuous-valued-treatment-effects","slug":"learning-continuous-valued-treatment-effects","title":"Using representation balancing to learn conditional-average dose responses from clustered data","date":"2023-09-07","arxiv_id":"2309.03731","repositories_listed":1,"syntology":null},{"url":"/paper/a-skeletonization-algorithm-for-gradient","slug":"a-skeletonization-algorithm-for-gradient","title":"A skeletonization algorithm for gradient-based optimization","date":"2023-09-05","arxiv_id":"2309.02527","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-skeletonization-algorithm-for-gradient#ran","syntology_url":"https://syntology.ai/paper/2309.02527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02527"}},"official":{"repos":["martinmenten/skeletonization-for-gradient-based-optimization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-in","slug":"benchmarking-large-language-models-in","title":"Benchmarking Large Language Models in Retrieval-Augmented Generation","date":"2023-09-04","arxiv_id":"2309.01431","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2309.01431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01431"}},"official":{"repos":["chen700564/RGB"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/transfer-learning-between-motor-imagery","slug":"transfer-learning-between-motor-imagery","title":"Transfer Learning between Motor Imagery Datasets using Deep Learning -- Validation of Framework and Comparison of Datasets","date":"2023-09-04","arxiv_id":"2311.16109","repositories_listed":1,"syntology":null},{"url":"/paper/turbulent-flow-simulation-using","slug":"turbulent-flow-simulation-using","title":"Benchmarking Autoregressive Conditional Diffusion Models for Turbulent Flow Simulation","date":"2023-09-04","arxiv_id":"2309.01745","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/turbulent-flow-simulation-using#ran","syntology_url":"https://syntology.ai/paper/2309.01745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01745"}},"official":{"repos":["tum-pbs/autoreg-pde-diffusion"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/orientation-independent-chinese-text","slug":"orientation-independent-chinese-text","title":"Orientation-Independent Chinese Text Recognition in Scene Images","date":"2023-09-03","arxiv_id":"2309.01081","repositories_listed":1,"syntology":null},{"url":"/paper/federatedscope-llm-a-comprehensive-package","slug":"federatedscope-llm-a-comprehensive-package","title":"FederatedScope-LLM: A Comprehensive Package for Fine-tuning Large Language Models in Federated Learning","date":"2023-09-01","arxiv_id":"2309.00363","repositories_listed":1,"syntology":null},{"url":"/paper/nemig-a-bilingual-news-collection-and","slug":"nemig-a-bilingual-news-collection-and","title":"NeMig -- A Bilingual News Collection and Knowledge Graph about Migration","date":"2023-09-01","arxiv_id":"2309.00550","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multilabel-topic-classification","slug":"benchmarking-multilabel-topic-classification","title":"Benchmarking Multilabel Topic Classification in the Kyrgyz Language","date":"2023-08-30","arxiv_id":"2308.15952","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-the-generation-of-fact-checking","slug":"benchmarking-the-generation-of-fact-checking","title":"Benchmarking the Generation of Fact Checking Explanations","date":"2023-08-29","arxiv_id":"2308.15202","repositories_listed":1,"syntology":null},{"url":"/paper/towards-quantitative-precision-for-ecg","slug":"towards-quantitative-precision-for-ecg","title":"Towards quantitative precision for ECG analysis: Leveraging state space models, self-supervision and patient metadata","date":"2023-08-29","arxiv_id":"2308.15291","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-quantitative-precision-for-ecg#ran","syntology_url":"https://syntology.ai/paper/2308.15291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.15291"}},"official":{"repos":["tmehari/ssm_ecg"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mllm-dataengine-an-iterative-refinement","slug":"mllm-dataengine-an-iterative-refinement","title":"MLLM-DataEngine: An Iterative Refinement Approach for MLLM","date":"2023-08-25","arxiv_id":"2308.13566","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mllm-dataengine-an-iterative-refinement#ran","syntology_url":"https://syntology.ai/paper/2308.13566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13566"}},"official":{"repos":["opendatalab/mllm-dataengine"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-document-page-classification-design","slug":"beyond-document-page-classification-design","title":"Beyond Document Page Classification: Design, Datasets, and Challenges","date":"2023-08-24","arxiv_id":"2308.12896","repositories_listed":1,"syntology":null},{"url":"/paper/finding-the-perfect-fit-applying-regression","slug":"finding-the-perfect-fit-applying-regression","title":"Finding the Perfect Fit: Applying Regression Models to ClimateBench v1.0","date":"2023-08-23","arxiv_id":"2308.11854","repositories_listed":1,"syntology":null},{"url":"/paper/llmrec-benchmarking-large-language-models-on","slug":"llmrec-benchmarking-large-language-models-on","title":"LLMRec: Benchmarking Large Language Models on Recommendation Task","date":"2023-08-23","arxiv_id":"2308.12241","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llmrec-benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2308.12241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12241"}},"official":{"repos":["williamliujl/llmrec"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/expecting-the-unexpected-towards-broad-out-of","slug":"expecting-the-unexpected-towards-broad-out-of","title":"Expecting The Unexpected: Towards Broad Out-Of-Distribution Detection","date":"2023-08-22","arxiv_id":"2308.11480","repositories_listed":1,"syntology":null},{"url":"/paper/multi-source-domain-adaptation-for-cross","slug":"multi-source-domain-adaptation-for-cross","title":"Benchmarking Domain Adaptation for Chemical Processes on the Tennessee Eastman Process","date":"2023-08-22","arxiv_id":"2308.11247","repositories_listed":1,"syntology":null},{"url":"/paper/xxmd-benchmarking-neural-force-fields-using","slug":"xxmd-benchmarking-neural-force-fields-using","title":"Beyond MD17: the reactive xxMD dataset","date":"2023-08-22","arxiv_id":"2308.11155","repositories_listed":1,"syntology":null},{"url":"/paper/ugsl-a-unified-framework-for-benchmarking","slug":"ugsl-a-unified-framework-for-benchmarking","title":"UGSL: A Unified Framework for Benchmarking Graph Structure Learning","date":"2023-08-21","arxiv_id":"2308.10737","repositories_listed":1,"syntology":null},{"url":"/paper/vi-net-boosting-category-level-6d-object-pose","slug":"vi-net-boosting-category-level-6d-object-pose","title":"VI-Net: Boosting Category-level 6D Object Pose Estimation via Learning Decoupled Rotations on the Spherical Representations","date":"2023-08-19","arxiv_id":"2308.09916","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vi-net-boosting-category-level-6d-object-pose#ran","syntology_url":"https://syntology.ai/paper/2308.09916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09916"}},"official":{"repos":["jiehonglin/vi-net"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/neurological-prognostication-of-post-cardiac","slug":"neurological-prognostication-of-post-cardiac","title":"Neurological Prognostication of Post-Cardiac-Arrest Coma Patients Using EEG Data: A Dynamic Survival Analysis Framework with Competing Risks","date":"2023-08-17","arxiv_id":"2308.11645","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-neural-network-generalization","slug":"benchmarking-neural-network-generalization","title":"Benchmarking Neural Network Generalization for Grammar Induction","date":"2023-08-16","arxiv_id":"2308.08253","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-scalable-epistemic-uncertainty","slug":"benchmarking-scalable-epistemic-uncertainty","title":"Benchmarking Scalable Epistemic Uncertainty Quantification in Organ Segmentation","date":"2023-08-15","arxiv_id":"2308.07506","repositories_listed":1,"syntology":null},{"url":"/paper/iot-data-trust-evaluation-via-machine","slug":"iot-data-trust-evaluation-via-machine","title":"IoT Data Trust Evaluation via Machine Learning","date":"2023-08-15","arxiv_id":"2308.11638","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-visual-analytics-framework-for","slug":"a-comparative-visual-analytics-framework-for","title":"A Comparative Visual Analytics Framework for Evaluating Evolutionary Processes in Multi-objective Optimization","date":"2023-08-10","arxiv_id":"2308.05640","repositories_listed":1,"syntology":null},{"url":"/paper/llmebench-a-flexible-framework-for","slug":"llmebench-a-flexible-framework-for","title":"LLMeBench: A Flexible Framework for Accelerating LLMs Benchmarking","date":"2023-08-09","arxiv_id":"2308.04945","repositories_listed":1,"syntology":null},{"url":"/paper/xflow-benchmarking-flow-behaviors-over-graphs","slug":"xflow-benchmarking-flow-behaviors-over-graphs","title":"XFlow: Benchmarking Flow Behaviors over Graphs","date":"2023-08-07","arxiv_id":"2308.03819","repositories_listed":1,"syntology":null},{"url":"/paper/precise-benchmarking-of-explainable-ai","slug":"precise-benchmarking-of-explainable-ai","title":"Precise Benchmarking of Explainable AI Attribution Methods","date":"2023-08-06","arxiv_id":"2308.03161","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/precise-benchmarking-of-explainable-ai#ran","syntology_url":"https://syntology.ai/paper/2308.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03161"}},"official":{"repos":["rbrandt1/precise-benchmarking-of-xai"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chatgpt-for-gtfs-from-words-to-information","slug":"chatgpt-for-gtfs-from-words-to-information","title":"ChatGPT for GTFS: Benchmarking LLMs on GTFS Understanding and Retrieval","date":"2023-08-04","arxiv_id":"2308.02618","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-ultra-high-definition-image-1","slug":"benchmarking-ultra-high-definition-image-1","title":"Benchmarking Ultra-High-Definition Image Reflection Removal","date":"2023-08-01","arxiv_id":"2308.00265","repositories_listed":1,"syntology":null},{"url":"/paper/qgym-a-gym-for-training-and-benchmarking-rl","slug":"qgym-a-gym-for-training-and-benchmarking-rl","title":"qgym: A Gym for Training and Benchmarking RL-Based Quantum Compilation","date":"2023-08-01","arxiv_id":"2308.02536","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-analyzing-robust-point-cloud","slug":"benchmarking-and-analyzing-robust-point-cloud","title":"Benchmarking and Analyzing Robust Point Cloud Recognition: Bag of Tricks for Defending Adversarial Examples","date":"2023-07-31","arxiv_id":"2307.16361","repositories_listed":1,"syntology":null},{"url":"/paper/visual-geo-localization-with-self-supervised","slug":"visual-geo-localization-with-self-supervised","title":"VG-SSL: Benchmarking Self-supervised Representation Learning Approaches for Visual Geo-localization","date":"2023-07-31","arxiv_id":"2308.00090","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-uncertainly-missing-and-ambiguous","slug":"rethinking-uncertainly-missing-and-ambiguous","title":"Rethinking Uncertainly Missing and Ambiguous Visual Modality in Multi-Modal Entity Alignment","date":"2023-07-30","arxiv_id":"2307.16210","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-uncertainly-missing-and-ambiguous#ran","syntology_url":"https://syntology.ai/paper/2307.16210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16210"}},"official":null}},{"url":"/paper/tmpnn-high-order-polynomial-regression-based","slug":"tmpnn-high-order-polynomial-regression-based","title":"TMPNN: High-Order Polynomial Regression Based on Taylor Map Factorization","date":"2023-07-30","arxiv_id":"2307.16105","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-anomaly-detection-system-on","slug":"benchmarking-anomaly-detection-system-on","title":"Benchmarking Jetson Edge Devices with an End-to-end Video-based Anomaly Detection System","date":"2023-07-28","arxiv_id":"2307.16834","repositories_listed":1,"syntology":null},{"url":"/paper/iml-vit-image-manipulation-localization-by","slug":"iml-vit-image-manipulation-localization-by","title":"IML-ViT: Benchmarking Image Manipulation Localization by Vision Transformer","date":"2023-07-27","arxiv_id":"2307.14863","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iml-vit-image-manipulation-localization-by#ran","syntology_url":"https://syntology.ai/paper/2307.14863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14863"}},"official":{"repos":["sunnyhaze/iml-vit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantitative-metrics-for-benchmarking-human","slug":"quantitative-metrics-for-benchmarking-human","title":"Quantitative Metrics for Benchmarking Human-Aware Robot Navigation","date":"2023-07-26","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/foundational-models-defining-a-new-era-in","slug":"foundational-models-defining-a-new-era-in","title":"Foundational Models Defining a New Era in Vision: A Survey and Outlook","date":"2023-07-25","arxiv_id":"2307.13721","repositories_listed":1,"syntology":null},{"url":"/paper/when-multi-task-learning-meets-partial","slug":"when-multi-task-learning-meets-partial","title":"When Multi-Task Learning Meets Partial Supervision: A Computer Vision Review","date":"2023-07-25","arxiv_id":"2307.14382","repositories_listed":1,"syntology":null},{"url":"/paper/remote-bio-sensing-open-source-benchmark","slug":"remote-bio-sensing-open-source-benchmark","title":"Remote Bio-Sensing: Open Source Benchmark Framework for Fair Evaluation of rPPG","date":"2023-07-24","arxiv_id":"2307.12644","repositories_listed":1,"syntology":null},{"url":"/paper/plantain-diffusion-inspired-pose-score","slug":"plantain-diffusion-inspired-pose-score","title":"PLANTAIN: Diffusion-inspired Pose Score Minimization for Fast and Accurate Molecular Docking","date":"2023-07-22","arxiv_id":"2307.12090","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plantain-diffusion-inspired-pose-score#ran","syntology_url":"https://syntology.ai/paper/2307.12090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12090"}},"official":{"repos":["molecularmodelinglab/plantain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/joingym-an-efficient-query-optimization","slug":"joingym-an-efficient-query-optimization","title":"JoinGym: An Efficient Query Optimization Environment for Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11704","repositories_listed":1,"syntology":null},{"url":"/paper/lala-shakti-swarup-ray-bo-zhou-sungho-suh","slug":"lala-shakti-swarup-ray-bo-zhou-sungho-suh","title":"Selecting the motion ground truth for loose-fitting wearables: benchmarking optical MoCap methods","date":"2023-07-21","arxiv_id":"2307.11881","repositories_listed":1,"syntology":null},{"url":"/paper/decoding-the-enigma-benchmarking-humans-and","slug":"decoding-the-enigma-benchmarking-humans-and","title":"Decoding the Enigma: Benchmarking Humans and AIs on the Many Facets of Working Memory","date":"2023-07-20","arxiv_id":"2307.10768","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoding-the-enigma-benchmarking-humans-and#ran","syntology_url":"https://syntology.ai/paper/2307.10768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10768"}},"official":{"repos":["zhanglab-deepneurocoglab/worm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scibench-evaluating-college-level-scientific","slug":"scibench-evaluating-college-level-scientific","title":"SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models","date":"2023-07-20","arxiv_id":"2307.10635","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scibench-evaluating-college-level-scientific#ran","syntology_url":"https://syntology.ai/paper/2307.10635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10635"}},"official":{"repos":["mandyyyyii/scibench"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-potential-based-rewards-for","slug":"benchmarking-potential-based-rewards-for","title":"Benchmarking Potential Based Rewards for Learning Humanoid Locomotion","date":"2023-07-19","arxiv_id":"2307.10142","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-accurate-optimal-transport-with","slug":"efficient-and-accurate-optimal-transport-with","title":"Efficient and Accurate Optimal Transport with Mirror Descent and Conjugate Gradients","date":"2023-07-17","arxiv_id":"2307.08507","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-and-accurate-optimal-transport-with#ran","syntology_url":"https://syntology.ai/paper/2307.08507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08507"}},"official":{"repos":["adaptive-agents-lab/mdot-pncg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-prediction-of-peptide-self-assembly","slug":"efficient-prediction-of-peptide-self-assembly","title":"Efficient Prediction of Peptide Self-assembly through Sequential and Graphical Encoding","date":"2023-07-17","arxiv_id":"2307.09169","repositories_listed":1,"syntology":null},{"url":"/paper/examining-the-effects-of-degree-distribution","slug":"examining-the-effects-of-degree-distribution","title":"Examining the Effects of Degree Distribution and Homophily in Graph Learning Models","date":"2023-07-17","arxiv_id":"2307.08881","repositories_listed":1,"syntology":null},{"url":"/paper/herolt-benchmarking-heterogeneous-long-tailed","slug":"herolt-benchmarking-heterogeneous-long-tailed","title":"Towards Heterogeneous Long-tailed Learning: Benchmarking, Metrics, and Toolbox","date":"2023-07-17","arxiv_id":"2307.08235","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/herolt-benchmarking-heterogeneous-long-tailed#ran","syntology_url":"https://syntology.ai/paper/2307.08235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08235"}},"official":{"repos":["ssskj/herolt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/easytpp-towards-open-benchmarking-the","slug":"easytpp-towards-open-benchmarking-the","title":"EasyTPP: Towards Open Benchmarking Temporal Point Processes","date":"2023-07-16","arxiv_id":"2307.08097","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/easytpp-towards-open-benchmarking-the#ran","syntology_url":"https://syntology.ai/paper/2307.08097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08097"}},"official":{"repos":["ant-research/easytemporalpointprocess"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gastrovision-a-multi-class-endoscopy-image","slug":"gastrovision-a-multi-class-endoscopy-image","title":"GastroVision: A Multi-class Endoscopy Image Dataset for Computer Aided Gastrointestinal Disease Detection","date":"2023-07-16","arxiv_id":"2307.08140","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gastrovision-a-multi-class-endoscopy-image#ran","syntology_url":"https://syntology.ai/paper/2307.08140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08140"}},"official":{"repos":["debeshjha/gastrovision"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-dynamic-points-removal-benchmark-in-point","slug":"a-dynamic-points-removal-benchmark-in-point","title":"A Dynamic Points Removal Benchmark in Point Cloud Maps","date":"2023-07-14","arxiv_id":"2307.07260","repositories_listed":1,"syntology":null},{"url":"/paper/intelligraphs-datasets-for-benchmarking","slug":"intelligraphs-datasets-for-benchmarking","title":"IntelliGraphs: Datasets for Benchmarking Knowledge Graph Generation","date":"2023-07-13","arxiv_id":"2307.06698","repositories_listed":1,"syntology":null},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comprehensive-overview-of-large-language","slug":"a-comprehensive-overview-of-large-language","title":"A Comprehensive Overview of Large Language Models","date":"2023-07-12","arxiv_id":"2307.06435","repositories_listed":1,"syntology":null},{"url":"/paper/anuraset-a-dataset-for-benchmarking","slug":"anuraset-a-dataset-for-benchmarking","title":"AnuraSet: A dataset for benchmarking Neotropical anuran calls identification in passive acoustic monitoring","date":"2023-07-11","arxiv_id":"2307.06860","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-algorithms-for-federated-domain","slug":"benchmarking-algorithms-for-federated-domain","title":"Benchmarking Algorithms for Federated Domain Generalization","date":"2023-07-11","arxiv_id":"2307.04942","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-algorithms-for-federated-domain#ran","syntology_url":"https://syntology.ai/paper/2307.04942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04942"}},"official":{"repos":["inouye-lab/feddg_benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unraveling-the-age-estimation-puzzle","slug":"unraveling-the-age-estimation-puzzle","title":"A Call to Reflect on Evaluation Practices for Age Estimation: Comparative Analysis of the State-of-the-Art and a Unified Benchmark","date":"2023-07-10","arxiv_id":"2307.04570","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-test-time-adaptation-against","slug":"benchmarking-test-time-adaptation-against","title":"Benchmarking Test-Time Adaptation against Distribution Shifts in Image Classification","date":"2023-07-06","arxiv_id":"2307.03133","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-test-time-adaptation-against#ran","syntology_url":"https://syntology.ai/paper/2307.03133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03133"}},"official":{"repos":["yuyongcan/benchmark-tta"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/performance-modeling-of-data-storage-systems","slug":"performance-modeling-of-data-storage-systems","title":"Performance Modeling of Data Storage Systems using Generative Models","date":"2023-07-05","arxiv_id":"2307.02073","repositories_listed":1,"syntology":null},{"url":"/paper/climatelearn-benchmarking-machine-learning-1","slug":"climatelearn-benchmarking-machine-learning-1","title":"ClimateLearn: Benchmarking Machine Learning for Weather and Climate Modeling","date":"2023-07-04","arxiv_id":"2307.01909","repositories_listed":1,"syntology":null},{"url":"/paper/scenereplica-benchmarking-real-world-robot","slug":"scenereplica-benchmarking-real-world-robot","title":"SCENEREPLICA: Benchmarking Real-World Robot Manipulation by Creating Replicable Scenes","date":"2023-06-27","arxiv_id":"2306.15620","repositories_listed":1,"syntology":null},{"url":"/paper/shuttleset22-benchmarking-stroke-forecasting","slug":"shuttleset22-benchmarking-stroke-forecasting","title":"Benchmarking Stroke Forecasting with Stroke-Level Badminton Dataset","date":"2023-06-27","arxiv_id":"2306.15664","repositories_listed":1,"syntology":null},{"url":"/paper/my-boli-code-mixed-marathi-english-corpora","slug":"my-boli-code-mixed-marathi-english-corpora","title":"My Boli: Code-mixed Marathi-English Corpora, Pretrained Language Models and Evaluation Benchmarks","date":"2023-06-24","arxiv_id":"2306.14030","repositories_listed":1,"syntology":null},{"url":"/paper/can-llms-express-their-uncertainty-an","slug":"can-llms-express-their-uncertainty-an","title":"Can LLMs Express Their Uncertainty? An Empirical Evaluation of Confidence Elicitation in LLMs","date":"2023-06-22","arxiv_id":"2306.13063","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-llms-express-their-uncertainty-an#ran","syntology_url":"https://syntology.ai/paper/2306.13063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13063"}},"official":{"repos":["miaoxiong2320/llm-uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/optiforest-optimal-isolation-forest-for","slug":"optiforest-optimal-isolation-forest-for","title":"OptIForest: Optimal Isolation Forest for Anomaly Detection","date":"2023-06-22","arxiv_id":"2306.12703","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/optiforest-optimal-isolation-forest-for#ran","syntology_url":"https://syntology.ai/paper/2306.12703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12703"}},"official":{"repos":["xiagll/optiforest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-analyzing-3d-aware-image-1","slug":"benchmarking-and-analyzing-3d-aware-image-1","title":"Benchmarking and Analyzing 3D-aware Image Synthesis with a Modularized Codebase","date":"2023-06-21","arxiv_id":"2306.12423","repositories_listed":1,"syntology":null},{"url":"/paper/gadbench-revisiting-and-benchmarking-1","slug":"gadbench-revisiting-and-benchmarking-1","title":"GADBench: Revisiting and Benchmarking Supervised Graph Anomaly Detection","date":"2023-06-21","arxiv_id":"2306.12251","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-learning-from-label-proportions","slug":"lightweight-learning-from-label-proportions","title":"On-orbit model training for satellite imagery with label proportions","date":"2023-06-21","arxiv_id":"2306.12461","repositories_listed":1,"syntology":null},{"url":"/paper/towards-mitigating-spurious-correlations-in","slug":"towards-mitigating-spurious-correlations-in","title":"Challenges and Opportunities in Improving Worst-Group Generalization in Presence of Spurious Features","date":"2023-06-21","arxiv_id":"2306.11957","repositories_listed":1,"syntology":null},{"url":"/paper/visogender-a-dataset-for-benchmarking-gender","slug":"visogender-a-dataset-for-benchmarking-gender","title":"VisoGender: A dataset for benchmarking gender bias in image-text pronoun resolution","date":"2023-06-21","arxiv_id":"2306.12424","repositories_listed":1,"syntology":null},{"url":"/paper/a-systematic-survey-in-geometric-deep","slug":"a-systematic-survey-in-geometric-deep","title":"Geometric Deep Learning for Structure-Based Drug Design: A Survey","date":"2023-06-20","arxiv_id":"2306.11768","repositories_listed":1,"syntology":null},{"url":"/paper/did-the-models-understand-documents","slug":"did-the-models-understand-documents","title":"Did the Models Understand Documents? Benchmarking Models for Language Understanding in Document-Level Relation Extraction","date":"2023-06-20","arxiv_id":"2306.11386","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/did-the-models-understand-documents#ran","syntology_url":"https://syntology.ai/paper/2306.11386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11386"}},"official":{"repos":["hytn/docred-hwe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imp-marl-a-suite-of-environments-for-large","slug":"imp-marl-a-suite-of-environments-for-large","title":"IMP-MARL: a Suite of Environments for Large-scale Infrastructure Management Planning via MARL","date":"2023-06-20","arxiv_id":"2306.11551","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-causalassembly-generating-realistic","slug":"texttt-causalassembly-generating-realistic","title":"$\\texttt{causalAssembly}$: Generating Realistic Production Data for Benchmarking Causal Discovery","date":"2023-06-19","arxiv_id":"2306.10816","repositories_listed":1,"syntology":null},{"url":"/paper/using-motif-transitions-for-temporal-graph","slug":"using-motif-transitions-for-temporal-graph","title":"Using Motif Transitions for Temporal Graph Generation","date":"2023-06-19","arxiv_id":"2306.11190","repositories_listed":1,"syntology":null},{"url":"/paper/companykg-a-large-scale-heterogeneous-graph","slug":"companykg-a-large-scale-heterogeneous-graph","title":"CompanyKG: A Large-Scale Heterogeneous Graph for Company Similarity Quantification","date":"2023-06-18","arxiv_id":"2306.10649","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/companykg-a-large-scale-heterogeneous-graph#ran","syntology_url":"https://syntology.ai/paper/2306.10649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10649"}},"official":{"repos":["eqtpartners/companykg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-graph-neural-networks-for-link","slug":"evaluating-graph-neural-networks-for-link","title":"Evaluating Graph Neural Networks for Link Prediction: Current Pitfalls and New Benchmarking","date":"2023-06-18","arxiv_id":"2306.10453","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-graph-neural-networks-for-link#ran","syntology_url":"https://syntology.ai/paper/2306.10453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10453"}},"official":{"repos":["juanhui28/heart"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/acoustic-identification-of-ae-aegypti","slug":"acoustic-identification-of-ae-aegypti","title":"Acoustic Identification of Ae. aegypti Mosquitoes using Smartphone Apps and Residual Convolutional Neural Networks","date":"2023-06-16","arxiv_id":"2306.10091","repositories_listed":1,"syntology":null},{"url":"/paper/are-large-language-models-really-good-logical","slug":"are-large-language-models-really-good-logical","title":"Are Large Language Models Really Good Logical Reasoners? A Comprehensive Evaluation and Beyond","date":"2023-06-16","arxiv_id":"2306.09841","repositories_listed":1,"syntology":null},{"url":"/paper/convolutional-and-deep-learning-based","slug":"convolutional-and-deep-learning-based","title":"Convolutional and Deep Learning based techniques for Time Series Ordinal Classification","date":"2023-06-16","arxiv_id":"2306.10084","repositories_listed":1,"syntology":null},{"url":"/paper/framework-and-benchmarks-for-combinatorial-1","slug":"framework-and-benchmarks-for-combinatorial-1","title":"Framework and Benchmarks for Combinatorial and Mixed-variable Bayesian Optimization","date":"2023-06-16","arxiv_id":"2306.09803","repositories_listed":1,"syntology":null}],"record_sha256":"502e5364d32742d02979fdab01204aab225031616f14559814863701ed6af688","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}