{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/10","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":56,"rows_per_page":100,"rows":[901,1000],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/9","next":"/task/benchmarking/papers/11","papers":[{"url":"/paper/asynchronous-batch-bayesian-optimization-with","slug":"asynchronous-batch-bayesian-optimization-with","title":"Asynchronous Batch Bayesian Optimization with Pipelining Evaluations for Experimental Resource$\\unicode{x2013}$constrained Conditions","date":"2024-12-05","arxiv_id":"2412.04392","repositories_listed":1,"syntology":null},{"url":"/paper/does-your-model-understand-genes-a-benchmark","slug":"does-your-model-understand-genes-a-benchmark","title":"Does your model understand genes? A benchmark of gene properties for biological and text models","date":"2024-12-05","arxiv_id":"2412.04075","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-descriptions-in-images-informs-zero","slug":"grounding-descriptions-in-images-informs-zero","title":"Grounding Descriptions in Images informs Zero-Shot Visual Recognition","date":"2024-12-05","arxiv_id":"2412.04429","repositories_listed":1,"syntology":null},{"url":"/paper/video-quality-assessment-a-comprehensive","slug":"video-quality-assessment-a-comprehensive","title":"Video Quality Assessment: A Comprehensive Survey","date":"2024-12-04","arxiv_id":"2412.04508","repositories_listed":1,"syntology":null},{"url":"/paper/bn-authprof-benchmarking-machine-learning-for","slug":"bn-authprof-benchmarking-machine-learning-for","title":"BN-AuthProf: Benchmarking Machine Learning for Bangla Author Profiling on Social Media Texts","date":"2024-12-03","arxiv_id":"2412.02058","repositories_listed":1,"syntology":null},{"url":"/paper/noisy-ostracods-a-fine-grained-imbalanced","slug":"noisy-ostracods-a-fine-grained-imbalanced","title":"Noisy Ostracods: A Fine-Grained, Imbalanced Real-World Dataset for Benchmarking Robust Machine Learning and Label Correction Methods","date":"2024-12-03","arxiv_id":"2412.02313","repositories_listed":1,"syntology":null},{"url":"/paper/prithvi-eo-2-0-a-versatile-multi-temporal","slug":"prithvi-eo-2-0-a-versatile-multi-temporal","title":"Prithvi-EO-2.0: A Versatile Multi-Temporal Foundation Model for Earth Observation Applications","date":"2024-12-03","arxiv_id":"2412.02732","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prithvi-eo-2-0-a-versatile-multi-temporal#ran","syntology_url":"https://syntology.ai/paper/2412.02732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02732"}},"official":{"repos":["NASA-IMPACT/Prithvi-EO-2.0"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentic-hls-an-agentic-reasoning-based-high","slug":"agentic-hls-an-agentic-reasoning-based-high","title":"Agentic-HLS: An agentic reasoning based high-level synthesis system using large language models (AI for EDA workshop 2024)","date":"2024-12-02","arxiv_id":"2412.01604","repositories_listed":1,"syntology":null},{"url":"/paper/commit0-library-generation-from-scratch","slug":"commit0-library-generation-from-scratch","title":"Commit0: Library Generation from Scratch","date":"2024-12-02","arxiv_id":"2412.01769","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/commit0-library-generation-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2412.01769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01769"}},"official":{"repos":["commit-0/commit0"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/down-with-the-hierarchy-the-h-in-hnsw-stands","slug":"down-with-the-hierarchy-the-h-in-hnsw-stands","title":"Down with the Hierarchy: The 'H' in HNSW Stands for \"Hubs\"","date":"2024-12-02","arxiv_id":"2412.01940","repositories_listed":1,"syntology":null},{"url":"/paper/textclass-benchmark-a-continuous-elo-rating","slug":"textclass-benchmark-a-continuous-elo-rating","title":"TextClass Benchmark: A Continuous Elo Rating of LLMs in Social Sciences","date":"2024-11-30","arxiv_id":"2412.00539","repositories_listed":1,"syntology":null},{"url":"/paper/circumventing-shortcuts-in-audio-visual","slug":"circumventing-shortcuts-in-audio-visual","title":"Circumventing shortcuts in audio-visual deepfake detection datasets with unsupervised learning","date":"2024-11-29","arxiv_id":"2412.00175","repositories_listed":1,"syntology":null},{"url":"/paper/openqdc-open-quantum-data-commons","slug":"openqdc-open-quantum-data-commons","title":"OpenQDC: Open Quantum Data Commons","date":"2024-11-29","arxiv_id":"2411.19629","repositories_listed":1,"syntology":null},{"url":"/paper/truth-or-mirage-towards-end-to-end-factuality","slug":"truth-or-mirage-towards-end-to-end-factuality","title":"Truth or Mirage? Towards End-to-End Factuality Evaluation with LLM-Oasis","date":"2024-11-29","arxiv_id":"2411.19655","repositories_listed":1,"syntology":null},{"url":"/paper/geobench-vlm-benchmarking-vision-language","slug":"geobench-vlm-benchmarking-vision-language","title":"GEOBench-VLM: Benchmarking Vision-Language Models for Geospatial Tasks","date":"2024-11-28","arxiv_id":"2411.19325","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/geobench-vlm-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.19325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19325"}},"official":{"repos":["the-ai-alliance/geo-bench-vlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coreval-a-comprehensive-and-objective","slug":"coreval-a-comprehensive-and-objective","title":"CHOICE: Benchmarking the Remote Sensing Capabilities of Large Vision-Language Models","date":"2024-11-27","arxiv_id":"2411.18145","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/coreval-a-comprehensive-and-objective#ran","syntology_url":"https://syntology.ai/paper/2411.18145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18145"}},"official":{"repos":["shawnan-whu/choice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/aigv-assessor-benchmarking-and-evaluating-the","slug":"aigv-assessor-benchmarking-and-evaluating-the","title":"AIGV-Assessor: Benchmarking and Evaluating the Perceptual Quality of Text-to-Video Generation with LMM","date":"2024-11-26","arxiv_id":"2411.17221","repositories_listed":1,"syntology":null},{"url":"/paper/machine-learning-for-the-digital-typhoon","slug":"machine-learning-for-the-digital-typhoon","title":"Machine Learning for the Digital Typhoon Dataset: Extensions to Multiple Basins and New Developments in Representations and Tasks","date":"2024-11-25","arxiv_id":"2411.16421","repositories_listed":1,"syntology":null},{"url":"/paper/vidhal-benchmarking-temporal-hallucinations","slug":"vidhal-benchmarking-temporal-hallucinations","title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","date":"2024-11-25","arxiv_id":"2411.16771","repositories_listed":1,"syntology":null},{"url":"/paper/reassessing-layer-pruning-in-llms-new","slug":"reassessing-layer-pruning-in-llms-new","title":"Reassessing Layer Pruning in LLMs: New Insights and Methods","date":"2024-11-23","arxiv_id":"2411.15558","repositories_listed":1,"syntology":null},{"url":"/paper/adamz-an-enhanced-optimisation-method-for","slug":"adamz-an-enhanced-optimisation-method-for","title":"AdamZ: An Enhanced Optimisation Method for Neural Network Training","date":"2024-11-22","arxiv_id":"2411.15375","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-the-robustness-of-optical-flow","slug":"benchmarking-the-robustness-of-optical-flow","title":"Benchmarking the Robustness of Optical Flow Estimation to Corruptions","date":"2024-11-22","arxiv_id":"2411.14865","repositories_listed":1,"syntology":null},{"url":"/paper/a-dataset-for-evaluating-online-anomaly","slug":"a-dataset-for-evaluating-online-anomaly","title":"PATH: A Discrete-sequence Dataset for Evaluating Online Unsupervised Anomaly Detection Approaches for Multivariate Time Series","date":"2024-11-21","arxiv_id":"2411.13951","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-gpt-4-against-human-translators","slug":"benchmarking-gpt-4-against-human-translators","title":"Benchmarking GPT-4 against Human Translators: A Comprehensive Evaluation Across Languages, Domains, and Expertise Levels","date":"2024-11-21","arxiv_id":"2411.13775","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-gpt-4-against-human-translators#ran","syntology_url":"https://syntology.ai/paper/2411.13775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13775"}},"official":{"repos":["elliottyan/gpt_versus_mt_experts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/forecasting-future-international-events-a","slug":"forecasting-future-international-events-a","title":"Forecasting Future International Events: A Reliable Dataset for Text-Based Event Modeling","date":"2024-11-21","arxiv_id":"2411.14042","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-environments-for-vehicle-routing","slug":"multi-agent-environments-for-vehicle-routing","title":"Multi-Agent Environments for Vehicle Routing Problems","date":"2024-11-21","arxiv_id":"2411.14411","repositories_listed":1,"syntology":null},{"url":"/paper/stackeval-benchmarking-llms-in-coding","slug":"stackeval-benchmarking-llms-in-coding","title":"StackEval: Benchmarking LLMs in Coding Assistance","date":"2024-11-21","arxiv_id":"2412.05288","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stackeval-benchmarking-llms-in-coding#ran","syntology_url":"https://syntology.ai/paper/2412.05288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05288"}},"official":{"repos":["ProsusAI/stack-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/delta-influence-unlearning-poisons-via","slug":"delta-influence-unlearning-poisons-via","title":"Delta-Influence: Unlearning Poisons via Influence Functions","date":"2024-11-20","arxiv_id":"2411.13731","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/delta-influence-unlearning-poisons-via#ran","syntology_url":"https://syntology.ai/paper/2411.13731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13731"}},"official":{"repos":["andyisokay/delta-influence"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/vbench-comprehensive-and-versatile-benchmark","slug":"vbench-comprehensive-and-versatile-benchmark","title":"VBench++: Comprehensive and Versatile Benchmark Suite for Video Generative Models","date":"2024-11-20","arxiv_id":"2411.13503","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-positional-encodings-for-gnns","slug":"benchmarking-positional-encodings-for-gnns","title":"Benchmarking Positional Encodings for GNNs and Graph Transformers","date":"2024-11-19","arxiv_id":"2411.12732","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-positional-encodings-for-gnns#ran","syntology_url":"https://syntology.ai/paper/2411.12732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12732"}},"official":{"repos":["ETH-DISCO/Benchmarking-PEs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dlbacktrace-a-model-agnostic-explainability","slug":"dlbacktrace-a-model-agnostic-explainability","title":"DLBacktrace: A Model Agnostic Explainability for any Deep Learning Models","date":"2024-11-19","arxiv_id":"2411.12643","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-pre-trained-text-embedding","slug":"benchmarking-pre-trained-text-embedding","title":"Benchmarking pre-trained text embedding models in aligning built asset information","date":"2024-11-18","arxiv_id":"2411.12056","repositories_listed":1,"syntology":null},{"url":"/paper/introducing-milabench-benchmarking","slug":"introducing-milabench-benchmarking","title":"Introducing Milabench: Benchmarking Accelerators for AI","date":"2024-11-18","arxiv_id":"2411.11940","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-preferences-of-vision-language","slug":"quantifying-preferences-of-vision-language","title":"Value-Spectrum: Quantifying Preferences of Vision-Language Models via Value Decomposition in Social Media Contexts","date":"2024-11-18","arxiv_id":"2411.11479","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-comprehensive-benchmark-for","slug":"towards-a-comprehensive-benchmark-for","title":"Towards a Comprehensive Benchmark for Pathological Lymph Node Metastasis in Breast Cancer Sections","date":"2024-11-16","arxiv_id":"2411.10752","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-probabilistic-generative","slug":"a-survey-of-probabilistic-generative","title":"A survey of probabilistic generative frameworks for molecular simulations","date":"2024-11-14","arxiv_id":"2411.09388","repositories_listed":1,"syntology":null},{"url":"/paper/beard-benchmarking-the-adversarial-robustness","slug":"beard-benchmarking-the-adversarial-robustness","title":"BEARD: Benchmarking the Adversarial Robustness for Dataset Distillation","date":"2024-11-14","arxiv_id":"2411.09265","repositories_listed":1,"syntology":null},{"url":"/paper/caravan-multimet-extending-caravan-with","slug":"caravan-multimet-extending-caravan-with","title":"Caravan MultiMet: Extending Caravan with Multiple Weather Nowcasts and Forecasts","date":"2024-11-14","arxiv_id":"2411.09459","repositories_listed":1,"syntology":null},{"url":"/paper/anomaly-detection-in-large-scale-cloud","slug":"anomaly-detection-in-large-scale-cloud","title":"Anomaly Detection in Large-Scale Cloud Systems: An Industry Case and Dataset","date":"2024-11-13","arxiv_id":"2411.09047","repositories_listed":1,"syntology":null},{"url":"/paper/fm-ts-flow-matching-for-time-series","slug":"fm-ts-flow-matching-for-time-series","title":"FM-TS: Flow Matching for Time Series Generation","date":"2024-11-12","arxiv_id":"2411.07506","repositories_listed":1,"syntology":{"n":20,"n_ran":20,"n_constructed":0,"n_ran_checked":16,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":11,"n_pointer_only":20,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 3 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fm-ts-flow-matching-for-time-series#ran","syntology_url":"https://syntology.ai/paper/2411.07506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07506"}},"official":{"repos":["unites-lab/fmts"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/arctique-an-artificial-histopathological","slug":"arctique-an-artificial-histopathological","title":"Arctique: An artificial histopathological dataset unifying realism and controllability for uncertainty quantification","date":"2024-11-11","arxiv_id":"2411.07097","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llms-judgments-with-no-gold","slug":"benchmarking-llms-judgments-with-no-gold","title":"Benchmarking LLMs' Judgments with No Gold Standard","date":"2024-11-11","arxiv_id":"2411.07127","repositories_listed":1,"syntology":null},{"url":"/paper/general-geospatial-inference-with-a","slug":"general-geospatial-inference-with-a","title":"General Geospatial Inference with a Population Dynamics Foundation Model","date":"2024-11-11","arxiv_id":"2411.07207","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-or-global-context-understanding-on","slug":"retrieval-or-global-context-understanding-on","title":"Retrieval or Global Context Understanding? On Many-Shot In-Context Learning for Long-Context Evaluation","date":"2024-11-11","arxiv_id":"2411.07130","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retrieval-or-global-context-understanding-on#ran","syntology_url":"https://syntology.ai/paper/2411.07130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07130"}},"official":{"repos":["launchnlp/ManyICLBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-distributional-alignment-of","slug":"benchmarking-distributional-alignment-of","title":"Benchmarking Distributional Alignment of Large Language Models","date":"2024-11-08","arxiv_id":"2411.05403","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-distributional-alignment-of#ran","syntology_url":"https://syntology.ai/paper/2411.05403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05403"}},"official":{"repos":["nicolemeister/benchmarking-distributional-alignment"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hourvideo-1-hour-video-language-understanding","slug":"hourvideo-1-hour-video-language-understanding","title":"HourVideo: 1-Hour Video-Language Understanding","date":"2024-11-07","arxiv_id":"2411.04998","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hourvideo-1-hour-video-language-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.04998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04998"}},"official":{"repos":["keshik6/HourVideo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beemo-benchmark-of-expert-edited-machine","slug":"beemo-benchmark-of-expert-edited-machine","title":"Beemo: Benchmark of Expert-edited Machine-generated Outputs","date":"2024-11-06","arxiv_id":"2411.04032","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-vision-language-model-unlearning","slug":"benchmarking-vision-language-model-unlearning","title":"Benchmarking Vision Language Model Unlearning via Fictitious Facial Identity Dataset","date":"2024-11-05","arxiv_id":"2411.03554","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-vision-language-model-unlearning#ran","syntology_url":"https://syntology.ai/paper/2411.03554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03554"}},"official":{"repos":["safolab-wisc/fiubench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/interaction2code-how-far-are-we-from","slug":"interaction2code-how-far-are-we-from","title":"Interaction2Code: Benchmarking MLLM-based Interactive Webpage Code Generation from Interactive Prototyping","date":"2024-11-05","arxiv_id":"2411.03292","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interaction2code-how-far-are-we-from#ran","syntology_url":"https://syntology.ai/paper/2411.03292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03292"}},"official":{"repos":["webpai/interaction2code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-loss-of-context-awareness-in-general","slug":"on-the-loss-of-context-awareness-in-general","title":"On the Loss of Context-awareness in General Instruction Fine-tuning","date":"2024-11-05","arxiv_id":"2411.02688","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-loss-of-context-awareness-in-general#ran","syntology_url":"https://syntology.ai/paper/2411.02688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02688"}},"official":{"repos":["YihanWang617/context_awareness"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/benchmarking-vision-language-action-models-on","slug":"benchmarking-vision-language-action-models-on","title":"Benchmarking Vision, Language, & Action Models on Robotic Learning Tasks","date":"2024-11-04","arxiv_id":"2411.05821","repositories_listed":1,"syntology":null},{"url":"/paper/layerdag-a-layerwise-autoregressive-diffusion","slug":"layerdag-a-layerwise-autoregressive-diffusion","title":"LayerDAG: A Layerwise Autoregressive Diffusion Model for Directed Acyclic Graph Generation","date":"2024-11-04","arxiv_id":"2411.02322","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/layerdag-a-layerwise-autoregressive-diffusion#ran","syntology_url":"https://syntology.ai/paper/2411.02322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02322"}},"official":{"repos":["graph-com/layerdag"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/tablegpt2-a-large-multimodal-model-with","slug":"tablegpt2-a-large-multimodal-model-with","title":"TableGPT2: A Large Multimodal Model with Tabular Data Integration","date":"2024-11-04","arxiv_id":"2411.02059","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tablegpt2-a-large-multimodal-model-with#ran","syntology_url":"https://syntology.ai/paper/2411.02059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02059"}},"official":{"repos":["tablegpt/tablegpt-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/road-waymo-action-awareness-at-scale-for","slug":"road-waymo-action-awareness-at-scale-for","title":"ROAD-Waymo: Action Awareness at Scale for Autonomous Driving","date":"2024-11-03","arxiv_id":"2411.01683","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/road-waymo-action-awareness-at-scale-for#ran","syntology_url":"https://syntology.ai/paper/2411.01683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.01683"}},"official":{"repos":["salmank255/ROAD_Waymo_Baseline"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/feet-a-framework-for-evaluating-embedding","slug":"feet-a-framework-for-evaluating-embedding","title":"FEET: A Framework for Evaluating Embedding Techniques","date":"2024-11-02","arxiv_id":"2411.01322","repositories_listed":1,"syntology":null},{"url":"/paper/cityscape-adverse-benchmarking-robustness-of","slug":"cityscape-adverse-benchmarking-robustness-of","title":"Cityscape-Adverse: Benchmarking Robustness of Semantic Segmentation with Realistic Scene Modifications via Diffusion-Based Image Editing","date":"2024-11-01","arxiv_id":"2411.00425","repositories_listed":1,"syntology":null},{"url":"/paper/libmoe-a-library-for-comprehensive","slug":"libmoe-a-library-for-comprehensive","title":"LIBMoE: A Library for comprehensive benchmarking Mixture of Experts in Large Language Models","date":"2024-11-01","arxiv_id":"2411.00918","repositories_listed":1,"syntology":null},{"url":"/paper/mirflex-music-information-retrieval-feature","slug":"mirflex-music-information-retrieval-feature","title":"MIRFLEX: Music Information Retrieval Feature Library for Extraction","date":"2024-11-01","arxiv_id":"2411.00469","repositories_listed":1,"syntology":null},{"url":"/paper/allclear-a-comprehensive-dataset-and","slug":"allclear-a-comprehensive-dataset-and","title":"AllClear: A Comprehensive Dataset and Benchmark for Cloud Removal in Satellite Imagery","date":"2024-10-31","arxiv_id":"2410.23891","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/allclear-a-comprehensive-dataset-and#ran","syntology_url":"https://syntology.ai/paper/2410.23891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23891"}},"official":{"repos":["zhou-hangyu/allclear"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/androidlab-training-and-systematic","slug":"androidlab-training-and-systematic","title":"AndroidLab: Training and Systematic Benchmarking of Android Autonomous Agents","date":"2024-10-31","arxiv_id":"2410.24024","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/androidlab-training-and-systematic#ran","syntology_url":"https://syntology.ai/paper/2410.24024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24024"}},"official":{"repos":["THUDM/Android-Lab"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/cale-continuous-arcade-learning-environment","slug":"cale-continuous-arcade-learning-environment","title":"CALE: Continuous Arcade Learning Environment","date":"2024-10-31","arxiv_id":"2410.23810","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cale-continuous-arcade-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2410.23810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23810"}},"official":{"repos":["farama-foundation/arcade-learning-environment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/detectrl-benchmarking-llm-generated-text","slug":"detectrl-benchmarking-llm-generated-text","title":"DetectRL: Benchmarking LLM-Generated Text Detection in Real-World Scenarios","date":"2024-10-31","arxiv_id":"2410.23746","repositories_listed":1,"syntology":{"n":16,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":11,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":16,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/detectrl-benchmarking-llm-generated-text#ran","syntology_url":"https://syntology.ai/paper/2410.23746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23746"}},"official":{"repos":["nlp2ct/detectrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/emgbench-benchmarking-out-of-distribution","slug":"emgbench-benchmarking-out-of-distribution","title":"EMGBench: Benchmarking Out-of-Distribution Generalization and Adaptation for Electromyography","date":"2024-10-31","arxiv_id":"2410.23625","repositories_listed":1,"syntology":null},{"url":"/paper/ideabench-benchmarking-large-language-models","slug":"ideabench-benchmarking-large-language-models","title":"IdeaBench: Benchmarking Large Language Models for Research Idea Generation","date":"2024-10-31","arxiv_id":"2411.02429","repositories_listed":1,"syntology":null},{"url":"/paper/llm-inference-bench-inference-benchmarking-of","slug":"llm-inference-bench-inference-benchmarking-of","title":"LLM-Inference-Bench: Inference Benchmarking of Large Language Models on AI Accelerators","date":"2024-10-31","arxiv_id":"2411.00136","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-inference-bench-inference-benchmarking-of#ran","syntology_url":"https://syntology.ai/paper/2411.00136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00136"}},"official":{"repos":["argonne-lcf/llm-inference-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm4mat-bench-benchmarking-large-language","slug":"llm4mat-bench-benchmarking-large-language","title":"LLM4Mat-Bench: Benchmarking Large Language Models for Materials Property Prediction","date":"2024-10-31","arxiv_id":"2411.00177","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm4mat-bench-benchmarking-large-language#ran","syntology_url":"https://syntology.ai/paper/2411.00177","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00177"}},"official":{"repos":["vertaix/llm4mat-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pedestrian-trajectory-prediction-with-missing","slug":"pedestrian-trajectory-prediction-with-missing","title":"Pedestrian Trajectory Prediction with Missing Data: Datasets, Imputation, and Benchmarking","date":"2024-10-31","arxiv_id":"2411.00174","repositories_listed":1,"syntology":null},{"url":"/paper/xrdslam-a-flexible-and-modular-framework-for","slug":"xrdslam-a-flexible-and-modular-framework-for","title":"XRDSLAM: A Flexible and Modular Framework for Deep Learning based SLAM","date":"2024-10-31","arxiv_id":"2410.23690","repositories_listed":1,"syntology":null},{"url":"/paper/coral-benchmarking-multi-turn-conversational","slug":"coral-benchmarking-multi-turn-conversational","title":"CORAL: Benchmarking Multi-turn Conversational Retrieval-Augmentation Generation","date":"2024-10-30","arxiv_id":"2410.23090","repositories_listed":1,"syntology":null},{"url":"/paper/datarec-a-framework-for-standardizing","slug":"datarec-a-framework-for-standardizing","title":"DataRec: A Python Library for Standardized and Reproducible Data Management in Recommender Systems","date":"2024-10-30","arxiv_id":"2410.22972","repositories_listed":1,"syntology":null},{"url":"/paper/ncadapt-dynamic-adaptation-with-domain","slug":"ncadapt-dynamic-adaptation-with-domain","title":"NCAdapt: Dynamic adaptation with domain-specific Neural Cellular Automata for continual hippocampus segmentation","date":"2024-10-30","arxiv_id":"2410.23368","repositories_listed":1,"syntology":null},{"url":"/paper/survey-of-cultural-awareness-in-language","slug":"survey-of-cultural-awareness-in-language","title":"Survey of Cultural Awareness in Language Models: Text and Beyond","date":"2024-10-30","arxiv_id":"2411.00860","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-human-and-automated-prompting-in","slug":"benchmarking-human-and-automated-prompting-in","title":"Benchmarking Human and Automated Prompting in the Segment Anything Model","date":"2024-10-29","arxiv_id":"2410.22048","repositories_listed":1,"syntology":null},{"url":"/paper/image2struct-benchmarking-structure","slug":"image2struct-benchmarking-structure","title":"Image2Struct: Benchmarking Structure Extraction for Vision-Language Models","date":"2024-10-29","arxiv_id":"2410.22456","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/image2struct-benchmarking-structure#ran","syntology_url":"https://syntology.ai/paper/2410.22456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22456"}},"official":{"repos":["stanford-crfm/helm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/pc-gym-benchmark-environments-for-process","slug":"pc-gym-benchmark-environments-for-process","title":"PC-Gym: Benchmark Environments For Process Control Problems","date":"2024-10-29","arxiv_id":"2410.22093","repositories_listed":1,"syntology":null},{"url":"/paper/autobench-v-can-large-vision-language-models","slug":"autobench-v-can-large-vision-language-models","title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","date":"2024-10-28","arxiv_id":"2410.21259","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobench-v-can-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21259"}},"official":{"repos":["wad3birch/AutoBench-V"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/codes-benchmarking-coupled-ode-surrogates","slug":"codes-benchmarking-coupled-ode-surrogates","title":"CODES: Benchmarking Coupled ODE Surrogates","date":"2024-10-28","arxiv_id":"2410.20886","repositories_listed":1,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":19,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/codes-benchmarking-coupled-ode-surrogates#ran","syntology_url":"https://syntology.ai/paper/2410.20886","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20886"}},"official":{"repos":["robin-janssen/codes-benchmark"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/curate-benchmarking-personalised-alignment-of","slug":"curate-benchmarking-personalised-alignment-of","title":"CURATe: Benchmarking Personalised Alignment of Conversational AI Assistants","date":"2024-10-28","arxiv_id":"2410.21159","repositories_listed":1,"syntology":null},{"url":"/paper/llmcbench-benchmarking-large-language-model","slug":"llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","arxiv_id":"2410.21352","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llmcbench-benchmarking-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2410.21352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21352"}},"official":{"repos":["aboveparadise/llmcbench"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/newterm-benchmarking-real-time-new-terms-for","slug":"newterm-benchmarking-real-time-new-terms-for","title":"NewTerm: Benchmarking Real-Time New Terms for Large Language Models with Annual Updates","date":"2024-10-28","arxiv_id":"2410.20814","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/newterm-benchmarking-real-time-new-terms-for#ran","syntology_url":"https://syntology.ai/paper/2410.20814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20814"}},"official":{"repos":["hexuandeng/newterm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/odrl-a-benchmark-for-off-dynamics","slug":"odrl-a-benchmark-for-off-dynamics","title":"ODRL: A Benchmark for Off-Dynamics Reinforcement Learning","date":"2024-10-28","arxiv_id":"2410.20750","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/odrl-a-benchmark-for-off-dynamics#ran","syntology_url":"https://syntology.ai/paper/2410.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20750"}},"official":{"repos":["offdynamicsrl/off-dynamics-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-large-language-model-based-hyper","slug":"sequential-large-language-model-based-hyper","title":"Sequential Large Language Model-Based Hyper-parameter Optimization","date":"2024-10-27","arxiv_id":"2410.20302","repositories_listed":1,"syntology":null},{"url":"/paper/spicepilot-navigating-spice-code-generation","slug":"spicepilot-navigating-spice-code-generation","title":"SPICEPilot: Navigating SPICE Code Generation and Simulation with AI Guidance","date":"2024-10-27","arxiv_id":"2410.20553","repositories_listed":1,"syntology":null},{"url":"/paper/automir-effective-zero-shot-medical","slug":"automir-effective-zero-shot-medical","title":"AutoMIR: Effective Zero-Shot Medical Information Retrieval without Relevance Labels","date":"2024-10-26","arxiv_id":"2410.20050","repositories_listed":1,"syntology":null},{"url":"/paper/ogbench-benchmarking-offline-goal-conditioned","slug":"ogbench-benchmarking-offline-goal-conditioned","title":"OGBench: Benchmarking Offline Goal-Conditioned RL","date":"2024-10-26","arxiv_id":"2410.20092","repositories_listed":1,"syntology":null},{"url":"/paper/agentsense-benchmarking-social-intelligence","slug":"agentsense-benchmarking-social-intelligence","title":"AgentSense: Benchmarking Social Intelligence of Language Agents through Interactive Scenarios","date":"2024-10-25","arxiv_id":"2410.19346","repositories_listed":1,"syntology":null},{"url":"/paper/an-auditing-test-to-detect-behavioral-shift","slug":"an-auditing-test-to-detect-behavioral-shift","title":"An Auditing Test To Detect Behavioral Shift in Language Models","date":"2024-10-25","arxiv_id":"2410.19406","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-auditing-test-to-detect-behavioral-shift#ran","syntology_url":"https://syntology.ai/paper/2410.19406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19406"}},"official":{"repos":["richterleo/Auditing_Test_for_LMs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/coqpilot-a-plugin-for-llm-based-generation-of","slug":"coqpilot-a-plugin-for-llm-based-generation-of","title":"CoqPilot, a plugin for LLM-based generation of proofs","date":"2024-10-25","arxiv_id":"2410.19605","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-diffusions-for-neural-posterior","slug":"conditional-diffusions-for-neural-posterior","title":"Conditional diffusions for amortized neural posterior estimation","date":"2024-10-24","arxiv_id":"2410.19105","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-diffusions-for-neural-posterior#ran","syntology_url":"https://syntology.ai/paper/2410.19105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19105"}},"official":{"repos":["tianyucodings/cdiff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/open6dor-benchmarking-open-instruction-6-dof","slug":"open6dor-benchmarking-open-instruction-6-dof","title":"Open6DOR: Benchmarking Open-instruction 6-DoF Object Rearrangement and A VLM-based Approach","date":"2024-10-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/robust-watermarking-using-generative-priors","slug":"robust-watermarking-using-generative-priors","title":"Robust Watermarking Using Generative Priors Against Image Editing: From Benchmarking to Advances","date":"2024-10-24","arxiv_id":"2410.18775","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-watermarking-using-generative-priors#ran","syntology_url":"https://syntology.ai/paper/2410.18775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18775"}},"official":{"repos":["shilin-lu/vine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/towards-better-open-ended-text-generation-a","slug":"towards-better-open-ended-text-generation-a","title":"Towards Better Open-Ended Text Generation: A Multicriteria Evaluation Framework","date":"2024-10-24","arxiv_id":"2410.18653","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-foundation-models-on-exceptional","slug":"benchmarking-foundation-models-on-exceptional","title":"Benchmarking Foundation Models on Exceptional Cases: Dataset Creation and Validation","date":"2024-10-23","arxiv_id":"2410.18001","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-for-image","slug":"benchmarking-large-language-models-for-image","title":"Benchmarking Large Language Models for Image Classification of Marine Mammals","date":"2024-10-22","arxiv_id":"2410.19848","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-multi-scene-fire-and-smoke","slug":"benchmarking-multi-scene-fire-and-smoke","title":"Benchmarking Multi-Scene Fire and Smoke Detection","date":"2024-10-22","arxiv_id":"2410.16631","repositories_listed":1,"syntology":null},{"url":"/paper/isimed-a-framework-for-self-supervised","slug":"isimed-a-framework-for-self-supervised","title":"ISImed: A Framework for Self-Supervised Learning using Intrinsic Spatial Information in Medical Images","date":"2024-10-22","arxiv_id":"2410.16947","repositories_listed":1,"syntology":null},{"url":"/paper/voicebench-benchmarking-llm-based-voice","slug":"voicebench-benchmarking-llm-based-voice","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","date":"2024-10-22","arxiv_id":"2410.17196","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voicebench-benchmarking-llm-based-voice#ran","syntology_url":"https://syntology.ai/paper/2410.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17196"}},"official":{"repos":["matthewcym/voicebench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-pathology-foundation-models","slug":"benchmarking-pathology-foundation-models","title":"Benchmarking Pathology Foundation Models: Adaptation Strategies and Scenarios","date":"2024-10-21","arxiv_id":"2410.16038","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-pathology-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2410.16038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16038"}},"official":{"repos":["quiil/benchmarkingpathologyfoundationmodels"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/building-conformal-prediction-intervals-with","slug":"building-conformal-prediction-intervals-with","title":"Building Conformal Prediction Intervals with Approximate Message Passing","date":"2024-10-21","arxiv_id":"2410.16493","repositories_listed":1,"syntology":null}],"record_sha256":"b7465ff32c8ead5f320392167c9b5ee5925aad0acef9ddb8e3a490209b9db715","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}