{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/3","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":56,"rows_per_page":100,"rows":[201,300],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/2","next":"/task/benchmarking/papers/4","papers":[{"url":"/paper/multi-stream-cellular-test-time-adaptation-of","slug":"multi-stream-cellular-test-time-adaptation-of","title":"Multi-Stream Cellular Test-Time Adaptation of Real-Time Models Evolving in Dynamic Environments","date":"2024-04-27","arxiv_id":"2404.17930","repositories_listed":2,"syntology":null},{"url":"/paper/seed-bench-2-plus-benchmarking-multimodal","slug":"seed-bench-2-plus-benchmarking-multimodal","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","date":"2024-04-25","arxiv_id":"2404.16790","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seed-bench-2-plus-benchmarking-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.16790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16790"}},"official":{"repos":["ailab-cvc/seed-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-benchmark-vision-foundation-models-for","slug":"how-to-benchmark-vision-foundation-models-for","title":"How to Benchmark Vision Foundation Models for Semantic Segmentation?","date":"2024-04-18","arxiv_id":"2404.12172","repositories_listed":2,"syntology":null},{"url":"/paper/second-edition-frcsyn-challenge-at-cvpr-2024","slug":"second-edition-frcsyn-challenge-at-cvpr-2024","title":"Second Edition FRCSyn Challenge at CVPR 2024: Face Recognition Challenge in the Era of Synthetic Data","date":"2024-04-16","arxiv_id":"2404.10378","repositories_listed":2,"syntology":null},{"url":"/paper/implicit-multi-spectral-transformer-an","slug":"implicit-multi-spectral-transformer-an","title":"Implicit Multi-Spectral Transformer: An Lightweight and Effective Visible to Infrared Image Translation Model","date":"2024-04-10","arxiv_id":"2404.07072","repositories_listed":2,"syntology":null},{"url":"/paper/ev2gym-a-flexible-v2g-simulator-for-ev-smart","slug":"ev2gym-a-flexible-v2g-simulator-for-ev-smart","title":"EV2Gym: A Flexible V2G Simulator for EV Smart Charging Research and Benchmarking","date":"2024-04-02","arxiv_id":"2404.01849","repositories_listed":2,"syntology":null},{"url":"/paper/codes-natural-language-to-code-repository-via","slug":"codes-natural-language-to-code-repository-via","title":"CodeS: Natural Language to Code Repository via Multi-Layer Sketch","date":"2024-03-25","arxiv_id":"2403.16443","repositories_listed":2,"syntology":null},{"url":"/paper/erase-benchmarking-feature-selection-methods","slug":"erase-benchmarking-feature-selection-methods","title":"ERASE: Benchmarking Feature Selection Methods for Deep Recommender Systems","date":"2024-03-19","arxiv_id":"2403.12660","repositories_listed":2,"syntology":null},{"url":"/paper/real-iad-a-real-world-multi-view-dataset-for","slug":"real-iad-a-real-world-multi-view-dataset-for","title":"Real-IAD: A Real-World Multi-View Dataset for Benchmarking Versatile Industrial Anomaly Detection","date":"2024-03-19","arxiv_id":"2403.12580","repositories_listed":2,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/real-iad-a-real-world-multi-view-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2403.12580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12580"}},"official":null}},{"url":"/paper/align-and-distill-unifying-and-improving","slug":"align-and-distill-unifying-and-improving","title":"Align and Distill: Unifying and Improving Domain Adaptive Object Detection","date":"2024-03-18","arxiv_id":"2403.12029","repositories_listed":2,"syntology":null},{"url":"/paper/text-r-2-bench-benchmarking-the-robustness-of","slug":"text-r-2-bench-benchmarking-the-robustness-of","title":"$\\text{R}^2$-Bench: Benchmarking the Robustness of Referring Perception Models under Perturbations","date":"2024-03-07","arxiv_id":"2403.04924","repositories_listed":2,"syntology":null},{"url":"/paper/injecagent-benchmarking-indirect-prompt","slug":"injecagent-benchmarking-indirect-prompt","title":"InjecAgent: Benchmarking Indirect Prompt Injections in Tool-Integrated Large Language Model Agents","date":"2024-03-05","arxiv_id":"2403.02691","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/injecagent-benchmarking-indirect-prompt#ran","syntology_url":"https://syntology.ai/paper/2403.02691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02691"}},"official":{"repos":["uiuc-kang-lab/injecagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-benchmarking-of-asynchronous-multi","slug":"fast-benchmarking-of-asynchronous-multi","title":"Fast Benchmarking of Asynchronous Multi-Fidelity Optimization on Zero-Cost Benchmarks","date":"2024-03-04","arxiv_id":"2403.01888","repositories_listed":2,"syntology":null},{"url":"/paper/real-colon-a-dataset-for-developing-real","slug":"real-colon-a-dataset-for-developing-real","title":"REAL-Colon: A dataset for developing real-world AI applications in colonoscopy","date":"2024-03-04","arxiv_id":"2403.02163","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-uncertainty-disentanglement","slug":"benchmarking-uncertainty-disentanglement","title":"Benchmarking Uncertainty Disentanglement: Specialized Uncertainties for Specialized Tasks","date":"2024-02-29","arxiv_id":"2402.19460","repositories_listed":2,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-uncertainty-disentanglement#ran","syntology_url":"https://syntology.ai/paper/2402.19460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19460"}},"official":{"repos":["bmucsanyi/bud","bmucsanyi/untangle"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-retrieval-augmented-generation","slug":"benchmarking-retrieval-augmented-generation","title":"Benchmarking Retrieval-Augmented Generation for Medicine","date":"2024-02-20","arxiv_id":"2402.13178","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2402.13178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13178"}},"official":{"repos":["teddy-xionggz/medrag","teddy-xionggz/mirage"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analobench-benchmarking-the-identification-of","slug":"analobench-benchmarking-the-identification-of","title":"AnaloBench: Benchmarking the Identification of Abstract and Long-context Analogies","date":"2024-02-19","arxiv_id":"2402.12370","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/analobench-benchmarking-the-identification-of#ran","syntology_url":"https://syntology.ai/paper/2402.12370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12370"}},"official":{"repos":["jhu-clsp/analogical-reasoning","JHU-CLSP/AnaloBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/causalgym-benchmarking-causal","slug":"causalgym-benchmarking-causal","title":"CausalGym: Benchmarking causal interpretability methods on linguistic tasks","date":"2024-02-19","arxiv_id":"2402.12560","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causalgym-benchmarking-causal#ran","syntology_url":"https://syntology.ai/paper/2402.12560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12560"}},"official":{"repos":["aryamanarora/causalgym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/attacknet-enhancing-biometric-security-via","slug":"attacknet-enhancing-biometric-security-via","title":"AttackNet: Enhancing Biometric Security via Tailored Convolutional Neural Network Architectures for Liveness Detection","date":"2024-02-06","arxiv_id":"2402.03769","repositories_listed":2,"syntology":null},{"url":"/paper/i-think-therefore-i-am-awareness-in-large","slug":"i-think-therefore-i-am-awareness-in-large","title":"I Think, Therefore I am: Benchmarking Awareness of Large Language Models Using AwareBench","date":"2024-01-31","arxiv_id":"2401.17882","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/i-think-therefore-i-am-awareness-in-large#ran","syntology_url":"https://syntology.ai/paper/2401.17882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17882"}},"official":{"repos":["howiehwong/awareness-in-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/multihop-rag-benchmarking-retrieval-augmented","slug":"multihop-rag-benchmarking-retrieval-augmented","title":"MultiHop-RAG: Benchmarking Retrieval-Augmented Generation for Multi-Hop Queries","date":"2024-01-27","arxiv_id":"2401.15391","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multihop-rag-benchmarking-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2401.15391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15391"}},"official":{"repos":["yixuantt/MultiHop-RAG"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentboard-an-analytical-evaluation-board-of","slug":"agentboard-an-analytical-evaluation-board-of","title":"AgentBoard: An Analytical Evaluation Board of Multi-turn LLM Agents","date":"2024-01-24","arxiv_id":"2401.13178","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentboard-an-analytical-evaluation-board-of#ran","syntology_url":"https://syntology.ai/paper/2401.13178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13178"}},"official":{"repos":["hkust-nlp/agentboard"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chex-gpt-harnessing-large-language-models-for","slug":"chex-gpt-harnessing-large-language-models-for","title":"CheX-GPT: Harnessing Large Language Models for Enhanced Chest X-ray Report Labeling","date":"2024-01-21","arxiv_id":"2401.11505","repositories_listed":2,"syntology":null},{"url":"/paper/authorship-obfuscation-in-multilingual","slug":"authorship-obfuscation-in-multilingual","title":"Authorship Obfuscation in Multilingual Machine-Generated Text Detection","date":"2024-01-15","arxiv_id":"2401.07867","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-the-cow-with-the-topcow","slug":"benchmarking-the-cow-with-the-topcow","title":"Benchmarking the CoW with the TopCoW Challenge: Topology-Aware Anatomical Segmentation of the Circle of Willis for CTA and MRA","date":"2023-12-29","arxiv_id":"2312.17670","repositories_listed":2,"syntology":null},{"url":"/paper/falcon-feature-label-constrained-graph-net","slug":"falcon-feature-label-constrained-graph-net","title":"FALCON: Feature-Label Constrained Graph Net Collapse for Memory Efficient GNNs","date":"2023-12-27","arxiv_id":"2312.16542","repositories_listed":2,"syntology":null},{"url":"/paper/knowledge-enhanced-conditional-imputation-for","slug":"knowledge-enhanced-conditional-imputation-for","title":"Knowledge Enhanced Conditional Imputation for Healthcare Time-series","date":"2023-12-27","arxiv_id":"2312.16713","repositories_listed":2,"syntology":null},{"url":"/paper/rdf-star2vec-rdf-star-graph-embeddings-for","slug":"rdf-star2vec-rdf-star-graph-embeddings-for","title":"RDF-star2Vec: RDF-star Graph Embeddings for Data Mining","date":"2023-12-25","arxiv_id":"2312.15626","repositories_listed":2,"syntology":null},{"url":"/paper/am-radio-agglomerative-model-reduce-all","slug":"am-radio-agglomerative-model-reduce-all","title":"AM-RADIO: Agglomerative Vision Foundation Model -- Reduce All Domains Into One","date":"2023-12-10","arxiv_id":"2312.06709","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/am-radio-agglomerative-model-reduce-all#ran","syntology_url":"https://syntology.ai/paper/2312.06709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06709"}},"official":{"repos":["nvlabs/radio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/benchlmm-benchmarking-cross-style-visual","slug":"benchlmm-benchmarking-cross-style-visual","title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","date":"2023-12-05","arxiv_id":"2312.02896","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchlmm-benchmarking-cross-style-visual#ran","syntology_url":"https://syntology.ai/paper/2312.02896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02896"}},"official":{"repos":["aifeg/benchgpt","aifeg/benchlmm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seed-bench-2-benchmarking-multimodal-large","slug":"seed-bench-2-benchmarking-multimodal-large","title":"SEED-Bench-2: Benchmarking Multimodal Large Language Models","date":"2023-11-28","arxiv_id":"2311.17092","repositories_listed":2,"syntology":null},{"url":"/paper/labcat-locally-adaptive-bayesian-optimization","slug":"labcat-locally-adaptive-bayesian-optimization","title":"LABCAT: Locally adaptive Bayesian optimization using principal-component-aligned trust regions","date":"2023-11-19","arxiv_id":"2311.11328","repositories_listed":2,"syntology":null},{"url":"/paper/do-localization-methods-actually-localize","slug":"do-localization-methods-actually-localize","title":"Do Localization Methods Actually Localize Memorized Data in LLMs? A Tale of Two Benchmarks","date":"2023-11-15","arxiv_id":"2311.09060","repositories_listed":2,"syntology":{"n":15,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/do-localization-methods-actually-localize#ran","syntology_url":"https://syntology.ai/paper/2311.09060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09060"}},"official":{"repos":["terarachang/memdata","terarachang/mempi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/locomujoco-a-comprehensive-imitation-learning","slug":"locomujoco-a-comprehensive-imitation-learning","title":"LocoMuJoCo: A Comprehensive Imitation Learning Benchmark for Locomotion","date":"2023-11-04","arxiv_id":"2311.02496","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/locomujoco-a-comprehensive-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2311.02496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02496"}},"official":{"repos":["robfiras/loco-mujoco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/empot-partial-alignment-of-density-maps-and","slug":"empot-partial-alignment-of-density-maps-and","title":"EMPOT: partial alignment of density maps and rigid body fitting using unbalanced Gromov-Wasserstein divergence","date":"2023-11-01","arxiv_id":"2311.00850","repositories_listed":2,"syntology":null},{"url":"/paper/battle-of-the-backbones-a-large-scale","slug":"battle-of-the-backbones-a-large-scale","title":"Battle of the Backbones: A Large-Scale Comparison of Pretrained Models across Computer Vision Tasks","date":"2023-10-30","arxiv_id":"2310.19909","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/battle-of-the-backbones-a-large-scale#ran","syntology_url":"https://syntology.ai/paper/2310.19909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19909"}},"official":{"repos":["hsouri/battle-of-the-backbones"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/oodrobustbench-benchmarking-and-analyzing","slug":"oodrobustbench-benchmarking-and-analyzing","title":"OODRobustBench: a Benchmark and Large-Scale Analysis of Adversarial Robustness under Distribution Shift","date":"2023-10-19","arxiv_id":"2310.12793","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/oodrobustbench-benchmarking-and-analyzing#ran","syntology_url":"https://syntology.ai/paper/2310.12793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12793"}},"official":{"repos":["oodrobustbench/oodrobustbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gdl-ds-a-benchmark-for-geometric-deep","slug":"gdl-ds-a-benchmark-for-geometric-deep","title":"GeSS: Benchmarking Geometric Deep Learning under Scientific Applications with Distribution Shifts","date":"2023-10-12","arxiv_id":"2310.08677","repositories_listed":2,"syntology":null},{"url":"/paper/metabox-a-benchmark-platform-for-meta-black-1","slug":"metabox-a-benchmark-platform-for-meta-black-1","title":"MetaBox: A Benchmark Platform for Meta-Black-Box Optimization with Reinforcement Learning","date":"2023-10-12","arxiv_id":"2310.08252","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/metabox-a-benchmark-platform-for-meta-black-1#ran","syntology_url":"https://syntology.ai/paper/2310.08252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08252"}},"official":{"repos":["GMC-DRL/MetaBox"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/akfruityield-modular-benchmarking-and-video","slug":"akfruityield-modular-benchmarking-and-video","title":"AKFruitYield: Modular benchmarking and video analysis software for Azure Kinect cameras for fruit size and fruit yield estimation in apple orchards","date":"2023-10-06","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-large-language-models-as-ai","slug":"benchmarking-large-language-models-as-ai","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","date":"2023-10-05","arxiv_id":"2310.03302","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-large-language-models-as-ai#ran","syntology_url":"https://syntology.ai/paper/2310.03302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03302"}},"official":{"repos":["snap-stanford/mlagentbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/rolellm-benchmarking-eliciting-and-enhancing","slug":"rolellm-benchmarking-eliciting-and-enhancing","title":"RoleLLM: Benchmarking, Eliciting, and Enhancing Role-Playing Abilities of Large Language Models","date":"2023-10-01","arxiv_id":"2310.00746","repositories_listed":2,"syntology":null},{"url":"/paper/smpler-x-scaling-up-expressive-human-pose-and","slug":"smpler-x-scaling-up-expressive-human-pose-and","title":"SMPLer-X: Scaling Up Expressive Human Pose and Shape Estimation","date":"2023-09-29","arxiv_id":"2309.17448","repositories_listed":2,"syntology":null},{"url":"/paper/lagrangebench-a-lagrangian-fluid-mechanics-1","slug":"lagrangebench-a-lagrangian-fluid-mechanics-1","title":"LagrangeBench: A Lagrangian Fluid Mechanics Benchmarking Suite","date":"2023-09-28","arxiv_id":"2309.16342","repositories_listed":2,"syntology":{"n":27,"n_ran":23,"n_constructed":0,"n_ran_checked":17,"n_instrument":6,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":4,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lagrangebench-a-lagrangian-fluid-mechanics-1#ran","syntology_url":"https://syntology.ai/paper/2309.16342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16342"}},"official":{"repos":["tumaer/lagrangebench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/a-toolkit-for-reliable-benchmarking-and","slug":"a-toolkit-for-reliable-benchmarking-and","title":"A Toolkit for Reliable Benchmarking and Research in Multi-Objective Reinforcement Learning","date":"2023-09-26","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/developing-a-scalable-benchmark-for-assessing","slug":"developing-a-scalable-benchmark-for-assessing","title":"Developing a Scalable Benchmark for Assessing Large Language Models in Knowledge Graph Engineering","date":"2023-08-31","arxiv_id":"2308.16622","repositories_listed":2,"syntology":null},{"url":"/paper/topical-chat-towards-knowledge-grounded-open-1","slug":"topical-chat-towards-knowledge-grounded-open-1","title":"Topical-Chat: Towards Knowledge-Grounded Open-Domain Conversations","date":"2023-08-23","arxiv_id":"2308.11995","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-generated-poses-how-rational-is","slug":"benchmarking-generated-poses-how-rational-is","title":"Benchmarking Generated Poses: How Rational is Structure-based Drug Design with Generative Models?","date":"2023-08-14","arxiv_id":"2308.07413","repositories_listed":2,"syntology":null},{"url":"/paper/bolaa-benchmarking-and-orchestrating-llm","slug":"bolaa-benchmarking-and-orchestrating-llm","title":"BOLAA: Benchmarking and Orchestrating LLM-augmented Autonomous Agents","date":"2023-08-11","arxiv_id":"2308.05960","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bolaa-benchmarking-and-orchestrating-llm#ran","syntology_url":"https://syntology.ai/paper/2308.05960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05960"}},"official":{"repos":["salesforce/bolaa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dig-in-evaluating-disparities-in-image","slug":"dig-in-evaluating-disparities-in-image","title":"DIG In: Evaluating Disparities in Image Generations with Indicators for Geographic Diversity","date":"2023-08-11","arxiv_id":"2308.06198","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/dig-in-evaluating-disparities-in-image#ran","syntology_url":"https://syntology.ai/paper/2308.06198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06198"}},"official":{"repos":["facebookresearch/dig-in"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/application-oriented-benchmarking-of-quantum","slug":"application-oriented-benchmarking-of-quantum","title":"Application-Oriented Benchmarking of Quantum Generative Learning Using QUARK","date":"2023-08-08","arxiv_id":"2308.04082","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-offline-reinforcement-learning-1","slug":"benchmarking-offline-reinforcement-learning-1","title":"Benchmarking Offline Reinforcement Learning on Real-Robot Hardware","date":"2023-07-28","arxiv_id":"2307.15690","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/benchmarking-offline-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2307.15690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15690"}},"official":{"repos":["rr-learning/trifinger-rl-example","rr-learning/trifinger_rl_datasets"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/intercode-standardizing-and-benchmarking","slug":"intercode-standardizing-and-benchmarking","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","date":"2023-06-26","arxiv_id":"2306.14898","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intercode-standardizing-and-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2306.14898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14898"}},"official":{"repos":["princeton-nlp/intercode"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-normal-on-the-evaluation-of-mutual-1","slug":"beyond-normal-on-the-evaluation-of-mutual-1","title":"Beyond Normal: On the Evaluation of Mutual Information Estimators","date":"2023-06-19","arxiv_id":"2306.11078","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-normal-on-the-evaluation-of-mutual-1#ran","syntology_url":"https://syntology.ai/paper/2306.11078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11078"}},"official":{"repos":["cbg-ethz/bmi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/opendataval-a-unified-benchmark-for-data-1","slug":"opendataval-a-unified-benchmark-for-data-1","title":"OpenDataVal: a Unified Benchmark for Data Valuation","date":"2023-06-18","arxiv_id":"2306.10577","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/opendataval-a-unified-benchmark-for-data-1#ran","syntology_url":"https://syntology.ai/paper/2306.10577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10577"}},"official":{"repos":["opendataval/opendataval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pinnacle-a-comprehensive-benchmark-of-physics","slug":"pinnacle-a-comprehensive-benchmark-of-physics","title":"PINNacle: A Comprehensive Benchmark of Physics-Informed Neural Networks for Solving PDEs","date":"2023-06-15","arxiv_id":"2306.08827","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/pinnacle-a-comprehensive-benchmark-of-physics#ran","syntology_url":"https://syntology.ai/paper/2306.08827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08827"}},"official":{"repos":["i207m/pinnacle"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/scale-scaling-up-the-complexity-for-advanced","slug":"scale-scaling-up-the-complexity-for-advanced","title":"One Law, Many Languages: Benchmarking Multilingual Legal Reasoning for Judicial Support","date":"2023-06-15","arxiv_id":"2306.09237","repositories_listed":2,"syntology":null},{"url":"/paper/muben-benchmarking-the-uncertainty-of-pre","slug":"muben-benchmarking-the-uncertainty-of-pre","title":"MUBen: Benchmarking the Uncertainty of Molecular Representation Models","date":"2023-06-14","arxiv_id":"2306.10060","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/muben-benchmarking-the-uncertainty-of-pre#ran","syntology_url":"https://syntology.ai/paper/2306.10060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10060"}},"official":{"repos":["Yinghao-Li/UncertaintyBenchmark","yinghao-li/muben"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/check-me-if-you-can-detecting-chatgpt","slug":"check-me-if-you-can-detecting-chatgpt","title":"On the Detectability of ChatGPT Content: Benchmarking, Methodology, and Evaluation through the Lens of Academic Writing","date":"2023-06-07","arxiv_id":"2306.05524","repositories_listed":2,"syntology":null},{"url":"/paper/libero-benchmarking-knowledge-transfer-for","slug":"libero-benchmarking-knowledge-transfer-for","title":"LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning","date":"2023-06-05","arxiv_id":"2306.03310","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/libero-benchmarking-knowledge-transfer-for#ran","syntology_url":"https://syntology.ai/paper/2306.03310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03310"}},"official":null}},{"url":"/paper/learning-from-integral-losses-in-physics","slug":"learning-from-integral-losses-in-physics","title":"Learning from Integral Losses in Physics Informed Neural Networks","date":"2023-05-27","arxiv_id":"2305.17387","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":1,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-integral-losses-in-physics#ran","syntology_url":"https://syntology.ai/paper/2305.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17387"}},"official":{"repos":["ehsansaleh/btspinn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/2305-14516","slug":"2305-14516","title":"Chakra: Advancing Performance Benchmarking and Co-design using Standardized Execution Traces","date":"2023-05-23","arxiv_id":"2305.14516","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2305-14516#ran","syntology_url":"https://syntology.ai/paper/2305.14516","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14516"}},"official":{"repos":["chakra-et/chakra"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/pmc-vqa-visual-instruction-tuning-for-medical","slug":"pmc-vqa-visual-instruction-tuning-for-medical","title":"PMC-VQA: Visual Instruction Tuning for Medical Visual Question Answering","date":"2023-05-17","arxiv_id":"2305.10415","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmc-vqa-visual-instruction-tuning-for-medical#ran","syntology_url":"https://syntology.ai/paper/2305.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10415"}},"official":{"repos":["xiaoman-zhang/PMC-VQA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-platform-for-the-biomedical-application-of","slug":"a-platform-for-the-biomedical-application-of","title":"A Platform for the Biomedical Application of Large Language Models","date":"2023-05-10","arxiv_id":"2305.06488","repositories_listed":2,"syntology":null},{"url":"/paper/assessing-working-memory-capacity-of-chatgpt","slug":"assessing-working-memory-capacity-of-chatgpt","title":"Working Memory Capacity of ChatGPT: An Empirical Study","date":"2023-04-30","arxiv_id":"2305.03731","repositories_listed":2,"syntology":null},{"url":"/paper/dsec-mos-segment-any-moving-object-with","slug":"dsec-mos-segment-any-moving-object-with","title":"Event-Free Moving Object Segmentation from Moving Ego Vehicle","date":"2023-04-28","arxiv_id":"2305.00126","repositories_listed":2,"syntology":null},{"url":"/paper/openagi-when-llm-meets-domain-experts","slug":"openagi-when-llm-meets-domain-experts","title":"OpenAGI: When LLM Meets Domain Experts","date":"2023-04-10","arxiv_id":"2304.04370","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/openagi-when-llm-meets-domain-experts#ran","syntology_url":"https://syntology.ai/paper/2304.04370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.04370"}},"official":{"repos":["agiresearch/openagi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scandeval-a-benchmark-for-scandinavian","slug":"scandeval-a-benchmark-for-scandinavian","title":"ScandEval: A Benchmark for Scandinavian Natural Language Processing","date":"2023-04-03","arxiv_id":"2304.00906","repositories_listed":2,"syntology":null},{"url":"/paper/codegeex-a-pre-trained-model-for-code","slug":"codegeex-a-pre-trained-model-for-code","title":"CodeGeeX: A Pre-Trained Model for Code Generation with Multilingual Benchmarking on HumanEval-X","date":"2023-03-30","arxiv_id":"2303.17568","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/codegeex-a-pre-trained-model-for-code#ran","syntology_url":"https://syntology.ai/paper/2303.17568","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17568"}},"official":{"repos":["THUDM/CodeGeeX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/imagenet-e-benchmarking-neural-network","slug":"imagenet-e-benchmarking-neural-network","title":"ImageNet-E: Benchmarking Neural Network Robustness via Attribute Editing","date":"2023-03-30","arxiv_id":"2303.17096","repositories_listed":2,"syntology":null},{"url":"/paper/automated-deep-learning-segmentation-of-high","slug":"automated-deep-learning-segmentation-of-high","title":"Automated deep learning segmentation of high-resolution 7 T postmortem MRI for quantitative analysis of structure-pathology correlations in neurodegenerative diseases","date":"2023-03-21","arxiv_id":"2303.12237","repositories_listed":2,"syntology":null},{"url":"/paper/covid-19-event-extraction-from-twitter-via","slug":"covid-19-event-extraction-from-twitter-via","title":"COVID-19 event extraction from Twitter via extractive question answering with continuous prompts","date":"2023-03-19","arxiv_id":"2303.10659","repositories_listed":2,"syntology":null},{"url":"/paper/highly-accurate-quantum-chemical-property","slug":"highly-accurate-quantum-chemical-property","title":"Highly Accurate Quantum Chemical Property Prediction with Uni-Mol+","date":"2023-03-16","arxiv_id":"2303.16982","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/highly-accurate-quantum-chemical-property#ran","syntology_url":"https://syntology.ai/paper/2303.16982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16982"}},"official":{"repos":["dptech-corp/Uni-Mol"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/cospgd-a-unified-white-box-adversarial-attack","slug":"cospgd-a-unified-white-box-adversarial-attack","title":"CosPGD: an efficient white-box adversarial attack for pixel-wise prediction tasks","date":"2023-02-04","arxiv_id":"2302.02213","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cospgd-a-unified-white-box-adversarial-attack#ran","syntology_url":"https://syntology.ai/paper/2302.02213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02213"}},"official":{"repos":["shashankskagnihotri/adv-corrected-ddcat-cospgd","shashankskagnihotri/cospgd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/heterogeneous-datasets-for-federated-survival","slug":"heterogeneous-datasets-for-federated-survival","title":"Heterogeneous Datasets for Federated Survival Analysis Simulation","date":"2023-01-28","arxiv_id":"2301.12166","repositories_listed":2,"syntology":null},{"url":"/paper/the-cropandweed-dataset-a-multi-modal","slug":"the-cropandweed-dataset-a-multi-modal","title":"The CropAndWeed Dataset: A Multi-Modal Learning Approach for Efficient Crop and Weed Manipulation","date":"2023-01-06","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/codebench-a-neural-architecture-and-hardware","slug":"codebench-a-neural-architecture-and-hardware","title":"CODEBench: A Neural Architecture and Hardware Accelerator Co-Design Framework","date":"2022-12-07","arxiv_id":"2212.03965","repositories_listed":2,"syntology":null},{"url":"/paper/a-call-to-reflect-on-evaluation-practices-for","slug":"a-call-to-reflect-on-evaluation-practices-for","title":"A Call to Reflect on Evaluation Practices for Failure Detection in Image Classification","date":"2022-11-28","arxiv_id":"2211.15259","repositories_listed":2,"syntology":null},{"url":"/paper/esb-a-benchmark-for-multi-domain-end-to-end","slug":"esb-a-benchmark-for-multi-domain-end-to-end","title":"ESB: A Benchmark For Multi-Domain End-to-End Speech Recognition","date":"2022-10-24","arxiv_id":"2210.13352","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/esb-a-benchmark-for-multi-domain-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2210.13352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13352"}},"official":null}},{"url":"/paper/spikesim-an-end-to-end-compute-in-memory","slug":"spikesim-an-end-to-end-compute-in-memory","title":"SpikeSim: An end-to-end Compute-in-Memory Hardware Evaluation Tool for Benchmarking Spiking Neural Networks","date":"2022-10-24","arxiv_id":"2210.12899","repositories_listed":2,"syntology":null},{"url":"/paper/idna-abf-multi-scale-deep-biological-language","slug":"idna-abf-multi-scale-deep-biological-language","title":"iDNA-ABF: multi-scale deep biological language learning model for the interpretable prediction of DNA methylations","date":"2022-10-17","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/a-comprehensive-study-on-large-scale-graph","slug":"a-comprehensive-study-on-large-scale-graph","title":"A Comprehensive Study on Large-Scale Graph Training: Benchmarking and Rethinking","date":"2022-10-14","arxiv_id":"2210.07494","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-comprehensive-study-on-large-scale-graph#ran","syntology_url":"https://syntology.ai/paper/2210.07494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07494"}},"official":{"repos":["vita-group/large_scale_gcn_benchmarking"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/medfair-benchmarking-fairness-for-medical","slug":"medfair-benchmarking-fairness-for-medical","title":"MEDFAIR: Benchmarking Fairness for Medical Imaging","date":"2022-10-04","arxiv_id":"2210.01725","repositories_listed":2,"syntology":null},{"url":"/paper/dynamic-backbone-protein-ligand-structure","slug":"dynamic-backbone-protein-ligand-structure","title":"State-specific protein-ligand complex structure prediction with a multi-scale deep generative model","date":"2022-09-30","arxiv_id":"2209.15171","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-learning-efficiency-in-deep","slug":"benchmarking-learning-efficiency-in-deep","title":"Benchmarking Learning Efficiency in Deep Reservoir Computing","date":"2022-09-29","arxiv_id":"2210.02549","repositories_listed":2,"syntology":null},{"url":"/paper/ferret-a-framework-for-benchmarking","slug":"ferret-a-framework-for-benchmarking","title":"ferret: a Framework for Benchmarking Explainers on Transformers","date":"2022-08-02","arxiv_id":"2208.01575","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ferret-a-framework-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2208.01575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.01575"}},"official":{"repos":["g8a9/ferret"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/why-do-tree-based-models-still-outperform","slug":"why-do-tree-based-models-still-outperform","title":"Why do tree-based models still outperform deep learning on tabular data?","date":"2022-07-18","arxiv_id":"2207.08815","repositories_listed":2,"syntology":null},{"url":"/paper/daisyrec-2-0-benchmarking-recommendation-for","slug":"daisyrec-2-0-benchmarking-recommendation-for","title":"DaisyRec 2.0: Benchmarking Recommendation for Rigorous Evaluation","date":"2022-06-22","arxiv_id":"2206.10848","repositories_listed":2,"syntology":null},{"url":"/paper/openxai-towards-a-transparent-evaluation-of","slug":"openxai-towards-a-transparent-evaluation-of","title":"OpenXAI: Towards a Transparent Evaluation of Model Explanations","date":"2022-06-22","arxiv_id":"2206.11104","repositories_listed":2,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/openxai-towards-a-transparent-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2206.11104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11104"}},"official":{"repos":["ai4life-group/openxai"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-node-outlier-detection-on-graphs","slug":"benchmarking-node-outlier-detection-on-graphs","title":"BOND: Benchmarking Unsupervised Outlier Node Detection on Static Attributed Graphs","date":"2022-06-21","arxiv_id":"2206.10071","repositories_listed":2,"syntology":null},{"url":"/paper/benchmarking-constraint-inference-in-inverse","slug":"benchmarking-constraint-inference-in-inverse","title":"Benchmarking Constraint Inference in Inverse Reinforcement Learning","date":"2022-06-20","arxiv_id":"2206.09670","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-constraint-inference-in-inverse#ran","syntology_url":"https://syntology.ai/paper/2206.09670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.09670"}},"official":{"repos":["guiliang/cirl-benchmarks-public","guiliang/icrl-benchmarks-public"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/motley-benchmarking-heterogeneity-and","slug":"motley-benchmarking-heterogeneity-and","title":"Motley: Benchmarking Heterogeneity and Personalization in Federated Learning","date":"2022-06-18","arxiv_id":"2206.09262","repositories_listed":2,"syntology":null},{"url":"/paper/long-range-graph-benchmark","slug":"long-range-graph-benchmark","title":"Long Range Graph Benchmark","date":"2022-06-16","arxiv_id":"2206.08164","repositories_listed":2,"syntology":null},{"url":"/paper/recbole-2-0-towards-a-more-up-to-date","slug":"recbole-2-0-towards-a-more-up-to-date","title":"RecBole 2.0: Towards a More Up-to-Date Recommendation Library","date":"2022-06-15","arxiv_id":"2206.07351","repositories_listed":2,"syntology":null},{"url":"/paper/data-driven-denoising-of-accelerometer","slug":"data-driven-denoising-of-accelerometer","title":"Data-Driven Denoising of Stationary Accelerometer Signals","date":"2022-06-13","arxiv_id":"2206.05937","repositories_listed":2,"syntology":null},{"url":"/paper/codes-a-distribution-shift-benchmark-dataset","slug":"codes-a-distribution-shift-benchmark-dataset","title":"CodeS: Towards Code Model Generalization Under Distribution Shift","date":"2022-06-11","arxiv_id":"2206.05480","repositories_listed":2,"syntology":null},{"url":"/paper/challenges-and-opportunities-in-offline","slug":"challenges-and-opportunities-in-offline","title":"Challenges and Opportunities in Offline Reinforcement Learning from Visual Observations","date":"2022-06-09","arxiv_id":"2206.04779","repositories_listed":2,"syntology":null},{"url":"/paper/mimii-dg-sound-dataset-for-malfunctioning","slug":"mimii-dg-sound-dataset-for-malfunctioning","title":"MIMII DG: Sound Dataset for Malfunctioning Industrial Machine Investigation and Inspection for Domain Generalization Task","date":"2022-05-27","arxiv_id":"2205.13879","repositories_listed":2,"syntology":null},{"url":"/paper/optimizing-performance-of-federated-person-re","slug":"optimizing-performance-of-federated-person-re","title":"Optimizing Performance of Federated Person Re-identification: Benchmarking and Analysis","date":"2022-05-24","arxiv_id":"2205.12144","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/optimizing-performance-of-federated-person-re#ran","syntology_url":"https://syntology.ai/paper/2205.12144","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12144"}},"official":{"repos":["cap-ntu/FedReID","EasyFL-AI/EasyFL"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/genisp-neural-isp-for-low-light-machine","slug":"genisp-neural-isp-for-low-light-machine","title":"GenISP: Neural ISP for Low-Light Machine Cognition","date":"2022-05-07","arxiv_id":"2205.03688","repositories_listed":2,"syntology":null}],"record_sha256":"e467ff05a53ed394741295339cb01535b6a83dbaaafc78d5153ef98cdf37af11","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}