{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/17","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":56,"rows_per_page":100,"rows":[1601,1700],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/16","next":"/task/benchmarking/papers/18","papers":[{"url":"/paper/loglead-fast-and-integrated-log-loader","slug":"loglead-fast-and-integrated-log-loader","title":"LogLead -- Fast and Integrated Log Loader, Enhancer, and Anomaly Detector","date":"2023-11-20","arxiv_id":"2311.11809","repositories_listed":1,"syntology":null},{"url":"/paper/a-reevaluation-of-event-extraction-past","slug":"a-reevaluation-of-event-extraction-past","title":"TextEE: Benchmark, Reevaluation, Reflections, and Future Challenges in Event Extraction","date":"2023-11-16","arxiv_id":"2311.09562","repositories_listed":1,"syntology":null},{"url":"/paper/abspyramid-benchmarking-the-abstraction","slug":"abspyramid-benchmarking-the-abstraction","title":"AbsPyramid: Benchmarking the Abstraction Ability of Language Models with a Unified Entailment Graph","date":"2023-11-15","arxiv_id":"2311.09174","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-generation-and-evaluation","slug":"benchmarking-generation-and-evaluation","title":"Benchmarking Generation and Evaluation Capabilities of Large Language Models for Instruction Controllable Summarization","date":"2023-11-15","arxiv_id":"2311.09184","repositories_listed":1,"syntology":null},{"url":"/paper/magic-benchmarking-large-language-model","slug":"magic-benchmarking-large-language-model","title":"MAgIC: Investigation of Large Language Model Powered Multi-Agent in Cognition, Adaptability, Rationality and Collaboration","date":"2023-11-14","arxiv_id":"2311.08562","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/magic-benchmarking-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2311.08562","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08562"}},"official":{"repos":["cathyxl/magic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-using-distribution-based-compositionality","slug":"on-using-distribution-based-compositionality","title":"On Using Distribution-Based Compositionality Assessment to Evaluate Compositional Generalisation in Machine Translation","date":"2023-11-14","arxiv_id":"2311.08249","repositories_listed":1,"syntology":null},{"url":"/paper/combinatorial-optimization-with-policy","slug":"combinatorial-optimization-with-policy","title":"Combinatorial Optimization with Policy Adaptation using Latent Space Search","date":"2023-11-13","arxiv_id":"2311.13569","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/combinatorial-optimization-with-policy#ran","syntology_url":"https://syntology.ai/paper/2311.13569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13569"}},"official":{"repos":["instadeepai/compass"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/connecting-the-dots-graph-neural-network","slug":"connecting-the-dots-graph-neural-network","title":"Connecting the Dots: Graph Neural Network Powered Ensemble and Classification of Medical Images","date":"2023-11-13","arxiv_id":"2311.07321","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-and-benchmarking-predict-then","slug":"rethinking-and-benchmarking-predict-then","title":"Benchmarking PtO and PnO Methods in the Predictive Combinatorial Optimization Regime","date":"2023-11-13","arxiv_id":"2311.07633","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-and-benchmarking-predict-then#ran","syntology_url":"https://syntology.ai/paper/2311.07633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07633"}},"official":{"repos":["thinklab-sjtu/predictiveco-benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/flames-benchmarking-value-alignment-of","slug":"flames-benchmarking-value-alignment-of","title":"Flames: Benchmarking Value Alignment of LLMs in Chinese","date":"2023-11-12","arxiv_id":"2311.06899","repositories_listed":1,"syntology":null},{"url":"/paper/cloudeval-yaml-a-practical-benchmark-for","slug":"cloudeval-yaml-a-practical-benchmark-for","title":"CloudEval-YAML: A Practical Benchmark for Cloud Configuration Generation","date":"2023-11-10","arxiv_id":"2401.06786","repositories_listed":1,"syntology":null},{"url":"/paper/multiiot-towards-large-scale-multisensory","slug":"multiiot-towards-large-scale-multisensory","title":"MultiIoT: Benchmarking Machine Learning for the Internet of Things","date":"2023-11-10","arxiv_id":"2311.06217","repositories_listed":1,"syntology":null},{"url":"/paper/tencentllmeval-a-hierarchical-evaluation-of","slug":"tencentllmeval-a-hierarchical-evaluation-of","title":"TencentLLMEval: A Hierarchical Evaluation of Real-World Capabilities for Human-Aligned LLMs","date":"2023-11-09","arxiv_id":"2311.05374","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-summarization-and-evaluation","slug":"a-comprehensive-summarization-and-evaluation","title":"A Comprehensive Summarization and Evaluation of Feature Refinement Modules for CTR Prediction","date":"2023-11-08","arxiv_id":"2311.04625","repositories_listed":1,"syntology":null},{"url":"/paper/the-petshop-dataset-finding-causes-of","slug":"the-petshop-dataset-finding-causes-of","title":"The PetShop Dataset -- Finding Causes of Performance Issues across Microservices","date":"2023-11-08","arxiv_id":"2311.04806","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-petshop-dataset-finding-causes-of#ran","syntology_url":"https://syntology.ai/paper/2311.04806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04806"}},"official":{"repos":["amazon-science/petshop-root-cause-analysis"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-voraus-ad-dataset-for-anomaly-detection","slug":"the-voraus-ad-dataset-for-anomaly-detection","title":"The voraus-AD Dataset for Anomaly Detection in Robot Applications","date":"2023-11-08","arxiv_id":"2311.04765","repositories_listed":1,"syntology":null},{"url":"/paper/bilingual-corpus-mining-and-multistage-fine","slug":"bilingual-corpus-mining-and-multistage-fine","title":"Bilingual Corpus Mining and Multistage Fine-Tuning for Improving Machine Translation of Lecture Transcripts","date":"2023-11-07","arxiv_id":"2311.03696","repositories_listed":1,"syntology":null},{"url":"/paper/deeppatent2-a-large-scale-benchmarking-corpus","slug":"deeppatent2-a-large-scale-benchmarking-corpus","title":"DeepPatent2: A Large-Scale Benchmarking Corpus for Technical Drawing Understanding","date":"2023-11-07","arxiv_id":"2311.04098","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-geospatial-question-answering","slug":"benchmarking-geospatial-question-answering","title":"Benchmarking Geospatial Question Answering Engines using the Dataset GeoQuestions1089","date":"2023-11-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hopfield-enhanced-deep-neural-networks-for","slug":"hopfield-enhanced-deep-neural-networks-for","title":"Hopfield-Enhanced Deep Neural Networks for Artifact-Resilient Brain State Decoding","date":"2023-11-06","arxiv_id":"2311.03421","repositories_listed":1,"syntology":null},{"url":"/paper/digital-typhoon-long-term-satellite-image","slug":"digital-typhoon-long-term-satellite-image","title":"Digital Typhoon: Long-term Satellite Image Dataset for the Spatio-Temporal Modeling of Tropical Cyclones","date":"2023-11-05","arxiv_id":"2311.02665","repositories_listed":1,"syntology":null},{"url":"/paper/jrdb-traj-a-dataset-and-benchmark-for","slug":"jrdb-traj-a-dataset-and-benchmark-for","title":"JRDB-Traj: A Dataset and Benchmark for Trajectory Forecasting in Crowds","date":"2023-11-05","arxiv_id":"2311.02736","repositories_listed":1,"syntology":null},{"url":"/paper/fragxsitedti-revealing-responsible-segments","slug":"fragxsitedti-revealing-responsible-segments","title":"FragXsiteDTI: Revealing Responsible Segments in Drug-Target Interaction with Transformer-Driven Interpretation","date":"2023-11-04","arxiv_id":"2311.02326","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fragxsitedti-revealing-responsible-segments#ran","syntology_url":"https://syntology.ai/paper/2311.02326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02326"}},"official":{"repos":["yazdanimehdi/fragxsitedti"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/neuroevobench-benchmarking-evolutionary","slug":"neuroevobench-benchmarking-evolutionary","title":"NeuroEvoBench: Benchmarking Evolutionary Optimizers for Deep Learning Applications","date":"2023-11-04","arxiv_id":"2311.02394","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neuroevobench-benchmarking-evolutionary#ran","syntology_url":"https://syntology.ai/paper/2311.02394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02394"}},"official":{"repos":["neuroevobench/neuroevobench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/grounded-intuition-of-gpt-vision-s-abilities","slug":"grounded-intuition-of-gpt-vision-s-abilities","title":"Grounded Intuition of GPT-Vision's Abilities with Scientific Images","date":"2023-11-03","arxiv_id":"2311.02069","repositories_listed":1,"syntology":null},{"url":"/paper/multi-eup-the-multilingual-european","slug":"multi-eup-the-multilingual-european","title":"Multi-EuP: The Multilingual European Parliament Dataset for Analysis of Bias in Information Retrieval","date":"2023-11-03","arxiv_id":"2311.01870","repositories_listed":1,"syntology":null},{"url":"/paper/replicable-benchmarking-of-neural-machine","slug":"replicable-benchmarking-of-neural-machine","title":"Replicable Benchmarking of Neural Machine Translation (NMT) on Low-Resource Local Languages in Indonesia","date":"2023-11-02","arxiv_id":"2311.00998","repositories_listed":1,"syntology":null},{"url":"/paper/ultra-efficient-on-device-object-detection-on","slug":"ultra-efficient-on-device-object-detection-on","title":"Ultra-Efficient On-Device Object Detection on AI-Integrated Smart Glasses with TinyissimoYOLO","date":"2023-11-02","arxiv_id":"2311.01057","repositories_listed":1,"syntology":null},{"url":"/paper/in-search-of-lost-online-test-time-adaptation","slug":"in-search-of-lost-online-test-time-adaptation","title":"In Search of Lost Online Test-time Adaptation: A Survey","date":"2023-10-31","arxiv_id":"2310.20199","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":7,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/in-search-of-lost-online-test-time-adaptation#ran","syntology_url":"https://syntology.ai/paper/2310.20199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20199"}},"official":{"repos":["jo-wang/otta_vit_survey"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/what-s-in-my-big-data","slug":"what-s-in-my-big-data","title":"What's In My Big Data?","date":"2023-10-31","arxiv_id":"2310.20707","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-s-in-my-big-data#ran","syntology_url":"https://syntology.ai/paper/2310.20707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20707"}},"official":{"repos":["allenai/wimbd"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/re-evaluating-retrosynthesis-algorithms-with","slug":"re-evaluating-retrosynthesis-algorithms-with","title":"Re-evaluating Retrosynthesis Algorithms with Syntheseus","date":"2023-10-30","arxiv_id":"2310.19796","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-llp-methods-challenges-and","slug":"evaluating-llp-methods-challenges-and","title":"Evaluating LLP Methods: Challenges and Approaches","date":"2023-10-29","arxiv_id":"2310.19065","repositories_listed":1,"syntology":null},{"url":"/paper/benchmark-generation-framework-with","slug":"benchmark-generation-framework-with","title":"Benchmark Generation Framework with Customizable Distortions for Image Classifier Robustness","date":"2023-10-28","arxiv_id":"2310.18626","repositories_listed":1,"syntology":null},{"url":"/paper/opendmc-an-open-source-library-and","slug":"opendmc-an-open-source-library-and","title":"OpenDMC: An Open-Source Library and Performance Evaluation for Deep-learning-based Multi-frame Compression","date":"2023-10-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/making-the-end-user-a-priority-in","slug":"making-the-end-user-a-priority-in","title":"OrionBench: Benchmarking Time Series Generative Models in the Service of the End-User","date":"2023-10-26","arxiv_id":"2310.17748","repositories_listed":1,"syntology":null},{"url":"/paper/xfever-exploring-fact-verification-across","slug":"xfever-exploring-fact-verification-across","title":"XFEVER: Exploring Fact Verification across Languages","date":"2023-10-25","arxiv_id":"2310.16278","repositories_listed":1,"syntology":null},{"url":"/paper/bless-benchmarking-large-language-models-on","slug":"bless-benchmarking-large-language-models-on","title":"BLESS: Benchmarking Large Language Models on Sentence Simplification","date":"2023-10-24","arxiv_id":"2310.15773","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bless-benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2310.15773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15773"}},"official":{"repos":["zurichnlp/bless"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mlfmf-data-sets-for-machine-learning-for-1","slug":"mlfmf-data-sets-for-machine-learning-for-1","title":"MLFMF: Data Sets for Machine Learning for Mathematical Formalization","date":"2023-10-24","arxiv_id":"2310.16005","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mlfmf-data-sets-for-machine-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2310.16005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16005"}},"official":{"repos":["ul-fmf/mlfmf-data"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/crow-benchmarking-commonsense-reasoning-in","slug":"crow-benchmarking-commonsense-reasoning-in","title":"CRoW: Benchmarking Commonsense Reasoning in Real-World Tasks","date":"2023-10-23","arxiv_id":"2310.15239","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crow-benchmarking-commonsense-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2310.15239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15239"}},"official":{"repos":["mismayil/crow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/designbench-exploring-and-benchmarking-dall-e","slug":"designbench-exploring-and-benchmarking-dall-e","title":"DEsignBench: Exploring and Benchmarking DALL-E 3 for Imagining Visual Design","date":"2023-10-23","arxiv_id":"2310.15144","repositories_listed":1,"syntology":null},{"url":"/paper/xtsc-bench-quantitative-benchmarking-for","slug":"xtsc-bench-quantitative-benchmarking-for","title":"XTSC-Bench: Quantitative Benchmarking for Explainers on Time Series Classification","date":"2023-10-23","arxiv_id":"2310.14957","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-improving-text-to-sql","slug":"benchmarking-and-improving-text-to-sql","title":"Benchmarking and Improving Text-to-SQL Generation under Ambiguity","date":"2023-10-20","arxiv_id":"2310.13659","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-and-improving-text-to-sql#ran","syntology_url":"https://syntology.ai/paper/2310.13659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13659"}},"official":{"repos":["testzer0/ambiqt"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-sequential-visual-input","slug":"benchmarking-sequential-visual-input","title":"Benchmarking Sequential Visual Input Reasoning and Prediction in Multimodal Large Language Models","date":"2023-10-20","arxiv_id":"2310.13473","repositories_listed":1,"syntology":null},{"url":"/paper/fast-hyperboloid-decision-tree-algorithms","slug":"fast-hyperboloid-decision-tree-algorithms","title":"Fast hyperboloid decision tree algorithms","date":"2023-10-20","arxiv_id":"2310.13841","repositories_listed":1,"syntology":null},{"url":"/paper/multitude-large-scale-multilingual-machine","slug":"multitude-large-scale-multilingual-machine","title":"MULTITuDE: Large-Scale Multilingual Machine-Generated Text Detection Benchmark","date":"2023-10-20","arxiv_id":"2310.13606","repositories_listed":1,"syntology":null},{"url":"/paper/prompt-injection-attacks-and-defenses-in-llm","slug":"prompt-injection-attacks-and-defenses-in-llm","title":"Formalizing and Benchmarking Prompt Injection Attacks and Defenses","date":"2023-10-19","arxiv_id":"2310.12815","repositories_listed":1,"syntology":null},{"url":"/paper/invig-benchmarking-interactive-visual","slug":"invig-benchmarking-interactive-visual","title":"InViG: Benchmarking Interactive Visual Grounding with 500K Human-Robot Interactions","date":"2023-10-18","arxiv_id":"2310.12147","repositories_listed":1,"syntology":null},{"url":"/paper/object-aware-inversion-and-reassembly-for","slug":"object-aware-inversion-and-reassembly-for","title":"Object-aware Inversion and Reassembly for Image Editing","date":"2023-10-18","arxiv_id":"2310.12149","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-aware-inversion-and-reassembly-for#ran","syntology_url":"https://syntology.ai/paper/2310.12149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12149"}},"official":{"repos":["aim-uofa/OIR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/to-generate-or-not-safety-driven-unlearned","slug":"to-generate-or-not-safety-driven-unlearned","title":"To Generate or Not? Safety-Driven Unlearned Diffusion Models Are Still Easy To Generate Unsafe Images ... For Now","date":"2023-10-18","arxiv_id":"2310.11868","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/to-generate-or-not-safety-driven-unlearned#ran","syntology_url":"https://syntology.ai/paper/2310.11868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11868"}},"official":{"repos":["optml-group/diffusion-mu-attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unveiling-the-siren-s-song-towards-reliable","slug":"unveiling-the-siren-s-song-towards-reliable","title":"FactCHD: Benchmarking Fact-Conflicting Hallucination Detection","date":"2023-10-18","arxiv_id":"2310.12086","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unveiling-the-siren-s-song-towards-reliable#ran","syntology_url":"https://syntology.ai/paper/2310.12086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12086"}},"official":{"repos":["zjunlp/factchd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dialoguellm-context-and-emotion-knowledge","slug":"dialoguellm-context-and-emotion-knowledge","title":"DialogueLLM: Context and Emotion Knowledge-Tuned Large Language Models for Emotion Recognition in Conversations","date":"2023-10-17","arxiv_id":"2310.11374","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dialoguellm-context-and-emotion-knowledge#ran","syntology_url":"https://syntology.ai/paper/2310.11374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11374"}},"official":{"repos":["Dreamyao516/DialogueLLM"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evalcrafter-benchmarking-and-evaluating-large","slug":"evalcrafter-benchmarking-and-evaluating-large","title":"EvalCrafter: Benchmarking and Evaluating Large Video Generation Models","date":"2023-10-17","arxiv_id":"2310.11440","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/evalcrafter-benchmarking-and-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2310.11440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11440"}},"official":{"repos":["EvalCrafter/EvalCrafter"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/3dyoga90-a-hierarchical-video-dataset-for","slug":"3dyoga90-a-hierarchical-video-dataset-for","title":"3DYoga90: A Hierarchical Video Dataset for Yoga Pose Understanding","date":"2023-10-16","arxiv_id":"2310.10131","repositories_listed":1,"syntology":null},{"url":"/paper/trigo-benchmarking-formal-mathematical-proof","slug":"trigo-benchmarking-formal-mathematical-proof","title":"TRIGO: Benchmarking Formal Mathematical Proof Reduction for Generative Language Models","date":"2023-10-16","arxiv_id":"2310.10180","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/trigo-benchmarking-formal-mathematical-proof#ran","syntology_url":"https://syntology.ai/paper/2310.10180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10180"}},"official":{"repos":["menik1126/TRIGO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mirage-model-agnostic-graph-distillation-for","slug":"mirage-model-agnostic-graph-distillation-for","title":"Mirage: Model-Agnostic Graph Distillation for Graph Classification","date":"2023-10-14","arxiv_id":"2310.09486","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mirage-model-agnostic-graph-distillation-for#ran","syntology_url":"https://syntology.ai/paper/2310.09486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09486"}},"official":{"repos":["idea-iitd/mirage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/randomized-benchmarking-of-local-zeroth-order","slug":"randomized-benchmarking-of-local-zeroth-order","title":"Randomized Benchmarking of Local Zeroth-Order Optimizers for Variational Quantum Systems","date":"2023-10-14","arxiv_id":"2310.09468","repositories_listed":1,"syntology":null},{"url":"/paper/banglanlp-at-blp-2023-task-2-benchmarking","slug":"banglanlp-at-blp-2023-task-2-benchmarking","title":"BanglaNLP at BLP-2023 Task 2: Benchmarking different Transformer Models for Sentiment Analysis of Bangla Social Media Posts","date":"2023-10-13","arxiv_id":"2310.09238","repositories_listed":1,"syntology":null},{"url":"/paper/kelly-is-a-warm-person-joseph-is-a-role-model","slug":"kelly-is-a-warm-person-joseph-is-a-role-model","title":"\"Kelly is a Warm Person, Joseph is a Role Model\": Gender Biases in LLM-Generated Reference Letters","date":"2023-10-13","arxiv_id":"2310.09219","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kelly-is-a-warm-person-joseph-is-a-role-model#ran","syntology_url":"https://syntology.ai/paper/2310.09219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09219"}},"official":{"repos":["uclanlp/biases-llm-reference-letters"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pose-format-library-for-viewing-augmenting","slug":"pose-format-library-for-viewing-augmenting","title":"pose-format: Library for Viewing, Augmenting, and Handling .pose Files","date":"2023-10-13","arxiv_id":"2310.09066","repositories_listed":1,"syntology":null},{"url":"/paper/welfare-diplomacy-benchmarking-language-model","slug":"welfare-diplomacy-benchmarking-language-model","title":"Welfare Diplomacy: Benchmarking Language Model Cooperation","date":"2023-10-13","arxiv_id":"2310.08901","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/welfare-diplomacy-benchmarking-language-model#ran","syntology_url":"https://syntology.ai/paper/2310.08901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08901"}},"official":{"repos":["mukobi/welfare-diplomacy"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mcu-a-task-centric-framework-for-open-ended","slug":"mcu-a-task-centric-framework-for-open-ended","title":"Towards Evaluating Generalist Agents: An Automated Benchmark in Open World","date":"2023-10-12","arxiv_id":"2310.08367","repositories_listed":1,"syntology":{"n":19,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":19,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/mcu-a-task-centric-framework-for-open-ended#ran","syntology_url":"https://syntology.ai/paper/2310.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08367"}},"official":{"repos":["craftjarvis/mcu"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/octopus-embodied-vision-language-programmer#ran","syntology_url":"https://syntology.ai/paper/2310.08588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08588"}},"official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/criteria-a-new-benchmarking-paradigm-for","slug":"criteria-a-new-benchmarking-paradigm-for","title":"CRITERIA: a New Benchmarking Paradigm for Evaluating Trajectory Prediction Models for Autonomous Driving","date":"2023-10-11","arxiv_id":"2310.07794","repositories_listed":1,"syntology":null},{"url":"/paper/probts-a-unified-toolkit-to-probe-deep-time","slug":"probts-a-unified-toolkit-to-probe-deep-time","title":"ProbTS: Benchmarking Point and Distributional Forecasting across Diverse Prediction Horizons","date":"2023-10-11","arxiv_id":"2310.07446","repositories_listed":1,"syntology":null},{"url":"/paper/transformers-for-green-semantic-communication","slug":"transformers-for-green-semantic-communication","title":"Transformers for Green Semantic Communication: Less Energy, More Semantics","date":"2023-10-11","arxiv_id":"2310.07592","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-explaining-large-language","slug":"benchmarking-and-explaining-large-language","title":"Benchmarking and Explaining Large Language Model-based Code Generation: A Causality-Centric Approach","date":"2023-10-10","arxiv_id":"2310.06680","repositories_listed":1,"syntology":null},{"url":"/paper/best-les-benchmarking-stroke-lesion","slug":"best-les-benchmarking-stroke-lesion","title":"BeSt-LeS: Benchmarking Stroke Lesion Segmentation using Deep Supervision","date":"2023-10-10","arxiv_id":"2310.07060","repositories_listed":1,"syntology":null},{"url":"/paper/what-if-the-tv-was-off-examining","slug":"what-if-the-tv-was-off-examining","title":"What If the TV Was Off? Examining Counterfactual Reasoning Abilities of Multi-modal Language Models","date":"2023-10-10","arxiv_id":"2310.06627","repositories_listed":1,"syntology":null},{"url":"/paper/transcending-the-attention-paradigm-implicit","slug":"transcending-the-attention-paradigm-implicit","title":"Transcending the Attention Paradigm: Representation Learning from Geospatial Social Media Data","date":"2023-10-09","arxiv_id":"2310.05378","repositories_listed":1,"syntology":null},{"url":"/paper/are-personalized-stochastic-parrots-more","slug":"are-personalized-stochastic-parrots-more","title":"Are Personalized Stochastic Parrots More Dangerous? Evaluating Persona Biases in Dialogue Systems","date":"2023-10-08","arxiv_id":"2310.05280","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/are-personalized-stochastic-parrots-more#ran","syntology_url":"https://syntology.ai/paper/2310.05280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05280"}},"official":{"repos":["uclanlp/persona-biases"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/hi-guys-or-hi-folks-benchmarking-gender","slug":"hi-guys-or-hi-folks-benchmarking-gender","title":"Hi Guys or Hi Folks? Benchmarking Gender-Neutral Machine Translation with the GeNTE Corpus","date":"2023-10-08","arxiv_id":"2310.05294","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hi-guys-or-hi-folks-benchmarking-gender#ran","syntology_url":"https://syntology.ai/paper/2310.05294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05294"}},"official":{"repos":["hlt-mt/fbk-neutr-eval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simplifying-gnn-performance-with-low-rank","slug":"simplifying-gnn-performance-with-low-rank","title":"Simple GNNs with Low Rank Non-parametric Aggregators","date":"2023-10-08","arxiv_id":"2310.05250","repositories_listed":1,"syntology":null},{"url":"/paper/fingpt-instruction-tuning-benchmark-for-open","slug":"fingpt-instruction-tuning-benchmark-for-open","title":"FinGPT: Instruction Tuning Benchmark for Open-Source Large Language Models in Financial Datasets","date":"2023-10-07","arxiv_id":"2310.04793","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fingpt-instruction-tuning-benchmark-for-open#ran","syntology_url":"https://syntology.ai/paper/2310.04793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04793"}},"official":{"repos":["ai4finance-foundation/fingpt"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pepmlm-target-sequence-conditioned-generation","slug":"pepmlm-target-sequence-conditioned-generation","title":"PepMLM: Target Sequence-Conditioned Generation of Therapeutic Peptide Binders via Span Masked Language Modeling","date":"2023-10-05","arxiv_id":"2310.03842","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/pepmlm-target-sequence-conditioned-generation#ran","syntology_url":"https://syntology.ai/paper/2310.03842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03842"}},"official":{"repos":["programmablebio/pepmlm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/can-language-models-employ-the-socratic","slug":"can-language-models-employ-the-socratic","title":"Can Language Models Employ the Socratic Method? Experiments with Code Debugging","date":"2023-10-04","arxiv_id":"2310.03210","repositories_listed":1,"syntology":null},{"url":"/paper/fully-automatic-segmentation-of-gross-target","slug":"fully-automatic-segmentation-of-gross-target","title":"Fully Automatic Segmentation of Gross Target Volume and Organs-at-Risk for Radiotherapy Planning of Nasopharyngeal Carcinoma","date":"2023-10-04","arxiv_id":"2310.02972","repositories_listed":1,"syntology":null},{"url":"/paper/t-3-bench-benchmarking-current-progress-in","slug":"t-3-bench-benchmarking-current-progress-in","title":"T$^3$Bench: Benchmarking Current Progress in Text-to-3D Generation","date":"2023-10-04","arxiv_id":"2310.02977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/t-3-bench-benchmarking-current-progress-in#ran","syntology_url":"https://syntology.ai/paper/2310.02977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02977"}},"official":{"repos":["THU-LYJ-Lab/T3Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causaltime-realistically-generated-time","slug":"causaltime-realistically-generated-time","title":"CausalTime: Realistically Generated Time-series for Benchmarking of Causal Discovery","date":"2023-10-03","arxiv_id":"2310.01753","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/causaltime-realistically-generated-time#ran","syntology_url":"https://syntology.ai/paper/2310.01753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01753"}},"official":{"repos":["jarrycyx/unn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gnnx-bench-unravelling-the-utility-of","slug":"gnnx-bench-unravelling-the-utility-of","title":"GNNX-BENCH: Unravelling the Utility of Perturbation-based GNN Explainers through In-depth Benchmarking","date":"2023-10-03","arxiv_id":"2310.01794","repositories_listed":1,"syntology":null},{"url":"/paper/learning-quantum-processes-with-quantum","slug":"learning-quantum-processes-with-quantum","title":"Learning Quantum Processes with Quantum Statistical Queries","date":"2023-10-03","arxiv_id":"2310.02075","repositories_listed":1,"syntology":null},{"url":"/paper/pgdqn-preference-guided-deep-q-network","slug":"pgdqn-preference-guided-deep-q-network","title":"PGDQN: Preference-Guided Deep Q-Network","date":"2023-10-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-visual-scene-understanding","slug":"adaptive-visual-scene-understanding","title":"Adaptive Visual Scene Understanding: Incremental Scene Graph Generation","date":"2023-10-02","arxiv_id":"2310.01636","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptive-visual-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2310.01636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01636"}},"official":{"repos":["zhanglab-deepneurocoglab/csegg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/newsreclib-a-pytorch-lightning-library-for","slug":"newsreclib-a-pytorch-lightning-library-for","title":"NewsRecLib: A PyTorch-Lightning Library for Neural News Recommendation","date":"2023-10-02","arxiv_id":"2310.01146","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/newsreclib-a-pytorch-lightning-library-for#ran","syntology_url":"https://syntology.ai/paper/2310.01146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01146"}},"official":{"repos":["andreeaiana/newsreclib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tram-benchmarking-temporal-reasoning-for","slug":"tram-benchmarking-temporal-reasoning-for","title":"TRAM: Benchmarking Temporal Reasoning for Large Language Models","date":"2023-10-02","arxiv_id":"2310.00835","repositories_listed":1,"syntology":null},{"url":"/paper/who-is-chatgpt-benchmarking-llms","slug":"who-is-chatgpt-benchmarking-llms","title":"Who is ChatGPT? Benchmarking LLMs' Psychological Portrayal Using PsychoBench","date":"2023-10-02","arxiv_id":"2310.01386","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/who-is-chatgpt-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2310.01386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01386"}},"official":{"repos":["cuhk-arise/psychobench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/felm-benchmarking-factuality-evaluation-of","slug":"felm-benchmarking-factuality-evaluation-of","title":"FELM: Benchmarking Factuality Evaluation of Large Language Models","date":"2023-10-01","arxiv_id":"2310.00741","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-cognitive-biases-in-large","slug":"benchmarking-cognitive-biases-in-large","title":"Benchmarking Cognitive Biases in Large Language Models as Evaluators","date":"2023-09-29","arxiv_id":"2309.17012","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-cognitive-biases-in-large#ran","syntology_url":"https://syntology.ai/paper/2309.17012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17012"}},"official":{"repos":["minnesotanlp/cobbler"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fedaiot-a-federated-learning-benchmark-for","slug":"fedaiot-a-federated-learning-benchmark-for","title":"FedAIoT: A Federated Learning Benchmark for Artificial Intelligence of Things","date":"2023-09-29","arxiv_id":"2310.00109","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fedaiot-a-federated-learning-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2310.00109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00109"}},"official":{"repos":["aiot-mlsys-lab/fedaiot"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/g4satbench-benchmarking-and-advancing-sat","slug":"g4satbench-benchmarking-and-advancing-sat","title":"G4SATBench: Benchmarking and Advancing SAT Solving with Graph Neural Networks","date":"2023-09-29","arxiv_id":"2309.16941","repositories_listed":1,"syntology":null},{"url":"/paper/muse-gnn-learning-unified-gene-representation","slug":"muse-gnn-learning-unified-gene-representation","title":"MuSe-GNN: Learning Unified Gene Representation From Multimodal Biological Graph Data","date":"2023-09-29","arxiv_id":"2310.02275","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/muse-gnn-learning-unified-gene-representation#ran","syntology_url":"https://syntology.ai/paper/2310.02275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02275"}},"official":{"repos":["helloworldlty/muse-gnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/forb-a-flat-object-retrieval-benchmark-for-1","slug":"forb-a-flat-object-retrieval-benchmark-for-1","title":"FORB: A Flat Object Retrieval Benchmark for Universal Image Embedding","date":"2023-09-28","arxiv_id":"2309.16249","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forb-a-flat-object-retrieval-benchmark-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.16249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16249"}},"official":{"repos":["pxiangwu/forb"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-fathom-benchmarking-large-language-models","slug":"gpt-fathom-benchmarking-large-language-models","title":"GPT-Fathom: Benchmarking Large Language Models to Decipher the Evolutionary Path towards GPT-4 and Beyond","date":"2023-09-28","arxiv_id":"2309.16583","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gpt-fathom-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2309.16583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16583"}},"official":{"repos":["gpt-fathom/gpt-fathom"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/lawbench-benchmarking-legal-knowledge-of","slug":"lawbench-benchmarking-legal-knowledge-of","title":"LawBench: Benchmarking Legal Knowledge of Large Language Models","date":"2023-09-28","arxiv_id":"2309.16289","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lawbench-benchmarking-legal-knowledge-of#ran","syntology_url":"https://syntology.ai/paper/2309.16289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16289"}},"official":{"repos":["open-compass/lawbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-trickle-down-impact-of-reward-in","slug":"the-trickle-down-impact-of-reward-in","title":"The Trickle-down Impact of Reward (In-)consistency on RLHF","date":"2023-09-28","arxiv_id":"2309.16155","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-trickle-down-impact-of-reward-in#ran","syntology_url":"https://syntology.ai/paper/2309.16155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16155"}},"official":{"repos":["shadowkiller33/contrast-instruction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-content-driven-micro-video-recommendation","slug":"a-content-driven-micro-video-recommendation","title":"A Content-Driven Micro-Video Recommendation Dataset at Scale","date":"2023-09-27","arxiv_id":"2309.15379","repositories_listed":1,"syntology":null},{"url":"/paper/nlpbench-evaluating-large-language-models-on","slug":"nlpbench-evaluating-large-language-models-on","title":"NLPBench: Evaluating Large Language Models on Solving NLP Problems","date":"2023-09-27","arxiv_id":"2309.15630","repositories_listed":1,"syntology":null},{"url":"/paper/node-aligned-graph-to-graph-generation-for","slug":"node-aligned-graph-to-graph-generation-for","title":"Node-Aligned Graph-to-Graph (NAG2G): Elevating Template-Free Deep Learning Approaches in Single-Step Retrosynthesis","date":"2023-09-27","arxiv_id":"2309.15798","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/node-aligned-graph-to-graph-generation-for#ran","syntology_url":"https://syntology.ai/paper/2309.15798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15798"}},"official":{"repos":["dptech-corp/nag2g"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/oceanbench-the-sea-surface-height-edition-1","slug":"oceanbench-the-sea-surface-height-edition-1","title":"OceanBench: The Sea Surface Height Edition","date":"2023-09-27","arxiv_id":"2309.15599","repositories_listed":1,"syntology":null},{"url":"/paper/unified-long-term-time-series-forecasting","slug":"unified-long-term-time-series-forecasting","title":"Unified Long-Term Time-Series Forecasting Benchmark","date":"2023-09-27","arxiv_id":"2309.15946","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-local-robustness-of-high","slug":"benchmarking-local-robustness-of-high","title":"Benchmarking Local Robustness of High-Accuracy Binary Neural Networks for Enhanced Traffic Sign Recognition","date":"2023-09-25","arxiv_id":"2310.03033","repositories_listed":1,"syntology":null}],"record_sha256":"a4246bd5a412bf30aa267a8e189282163c6e43c312b8c4e12ab24e34c247983a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}