{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/16","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":16,"pages_in_order":56,"rows_per_page":100,"rows":[1501,1600],"of":5548,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking","prev":"/task/benchmarking/papers/15","next":"/task/benchmarking/papers/17","papers":[{"url":"/paper/explainable-global-wildfire-prediction-models","slug":"explainable-global-wildfire-prediction-models","title":"Explainable Global Wildfire Prediction Models using Graph Neural Networks","date":"2024-02-11","arxiv_id":"2402.07152","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-on-1","slug":"benchmarking-large-language-models-on-1","title":"Benchmarking Large Language Models on Communicative Medical Coaching: a Novel System and Dataset","date":"2024-02-08","arxiv_id":"2402.05547","repositories_listed":1,"syntology":null},{"url":"/paper/natural-language-user-profiles-for","slug":"natural-language-user-profiles-for","title":"Transparent and Scrutable Recommendations Using Natural Language User Profiles","date":"2024-02-08","arxiv_id":"2402.05810","repositories_listed":1,"syntology":null},{"url":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bri3l-a-brightness-illusion-image-dataset-for","slug":"bri3l-a-brightness-illusion-image-dataset-for","title":"BRI3L: A Brightness Illusion Image Dataset for Identification and Localization of Regions of Illusory Perception","date":"2024-02-07","arxiv_id":"2402.04541","repositories_listed":1,"syntology":null},{"url":"/paper/instructscene-instruction-driven-3d-indoor","slug":"instructscene-instruction-driven-3d-indoor","title":"InstructScene: Instruction-Driven 3D Indoor Scene Synthesis with Semantic Graph Prior","date":"2024-02-07","arxiv_id":"2402.04717","repositories_listed":1,"syntology":null},{"url":"/paper/on-diffusion-models-for-amortized-inference","slug":"on-diffusion-models-for-amortized-inference","title":"Improved off-policy training of diffusion samplers","date":"2024-02-07","arxiv_id":"2402.05098","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-diffusion-models-for-amortized-inference#ran","syntology_url":"https://syntology.ai/paper/2402.05098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05098"}},"official":{"repos":["gfnorg/gfn-diffusion"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-biologically-plausible-and-private","slug":"towards-biologically-plausible-and-private","title":"Towards Biologically Plausible and Private Gene Expression Data Generation","date":"2024-02-07","arxiv_id":"2402.04912","repositories_listed":1,"syntology":null},{"url":"/paper/ltu-ili-an-all-in-one-framework-for-implicit","slug":"ltu-ili-an-all-in-one-framework-for-implicit","title":"LtU-ILI: An All-in-One Framework for Implicit Inference in Astrophysics and Cosmology","date":"2024-02-06","arxiv_id":"2402.05137","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ltu-ili-an-all-in-one-framework-for-implicit#ran","syntology_url":"https://syntology.ai/paper/2402.05137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05137"}},"official":{"repos":["maho3/ltu-ili"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lv-eval-a-balanced-long-context-benchmark","slug":"lv-eval-a-balanced-long-context-benchmark","title":"LV-Eval: A Balanced Long-Context Benchmark with 5 Length Levels Up to 256K","date":"2024-02-06","arxiv_id":"2402.05136","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lv-eval-a-balanced-long-context-benchmark#ran","syntology_url":"https://syntology.ai/paper/2402.05136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05136"}},"official":{"repos":["infinigence/lveval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/ct-based-anatomical-segmentation-for-thoracic","slug":"ct-based-anatomical-segmentation-for-thoracic","title":"Architecture Analysis and Benchmarking of 3D U-shaped Deep Learning Models for Thoracic Anatomical Segmentation","date":"2024-02-05","arxiv_id":"2402.03230","repositories_listed":1,"syntology":null},{"url":"/paper/jobskape-a-framework-for-generating-synthetic","slug":"jobskape-a-framework-for-generating-synthetic","title":"JOBSKAPE: A Framework for Generating Synthetic Job Postings to Enhance Skill Matching","date":"2024-02-05","arxiv_id":"2402.03242","repositories_listed":1,"syntology":null},{"url":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/effibench-benchmarking-the-efficiency-of#ran","syntology_url":"https://syntology.ai/paper/2402.02037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02037"}},"official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/genface-a-large-scale-fine-grained-face","slug":"genface-a-large-scale-fine-grained-face","title":"GenFace: A Large-Scale Fine-Grained Face Forgery Benchmark and Cross Appearance-Edge Learning","date":"2024-02-03","arxiv_id":"2402.02003","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genface-a-large-scale-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2402.02003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02003"}},"official":{"repos":["jenine-321/genface"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/probing-critical-learning-dynamics-of-plms","slug":"probing-critical-learning-dynamics-of-plms","title":"Probing Critical Learning Dynamics of PLMs for Hate Speech Detection","date":"2024-02-03","arxiv_id":"2402.02144","repositories_listed":1,"syntology":null},{"url":"/paper/vi-e-va-llm-a-conceptual-stack-for-evaluating","slug":"vi-e-va-llm-a-conceptual-stack-for-evaluating","title":"Vi(E)va LLM! A Conceptual Stack for Evaluating and Interpreting Generative AI-based Visualizations","date":"2024-02-03","arxiv_id":"2402.02167","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-limitations-of-graph-reasoning","slug":"exploring-the-limitations-of-graph-reasoning","title":"Can LLMs perform structured graph reasoning?","date":"2024-02-02","arxiv_id":"2402.01805","repositories_listed":1,"syntology":null},{"url":"/paper/short-benchmarking-transferable-adversarial","slug":"short-benchmarking-transferable-adversarial","title":"Benchmarking Transferable Adversarial Attacks","date":"2024-02-01","arxiv_id":"2402.00418","repositories_listed":1,"syntology":null},{"url":"/paper/we-re-not-using-videos-effectively-an-updated","slug":"we-re-not-using-videos-effectively-an-updated","title":"We're Not Using Videos Effectively: An Updated Domain Adaptive Video Segmentation Baseline","date":"2024-02-01","arxiv_id":"2402.00868","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/we-re-not-using-videos-effectively-an-updated#ran","syntology_url":"https://syntology.ai/paper/2402.00868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00868"}},"official":{"repos":["simarkareer/unifiedvideoda"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-benchmarking-for-iterative","slug":"explainable-benchmarking-for-iterative","title":"Explainable Benchmarking for Iterative Optimization Heuristics","date":"2024-01-31","arxiv_id":"2401.17842","repositories_listed":1,"syntology":null},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/category-wise-fine-tuning-resisting-incorrect","slug":"category-wise-fine-tuning-resisting-incorrect","title":"Category-wise Fine-Tuning: Resisting Incorrect Pseudo-Labels in Multi-Label Image Classification with Partial Labels","date":"2024-01-30","arxiv_id":"2401.16991","repositories_listed":1,"syntology":null},{"url":"/paper/planning-creation-usage-benchmarking-llms-for","slug":"planning-creation-usage-benchmarking-llms-for","title":"Planning, Creation, Usage: Benchmarking LLMs for Comprehensive Tool Utilization in Real-World Complex Scenarios","date":"2024-01-30","arxiv_id":"2401.17167","repositories_listed":1,"syntology":null},{"url":"/paper/machine-translation-meta-evaluation-through","slug":"machine-translation-meta-evaluation-through","title":"Machine Translation Meta Evaluation through Translation Accuracy Challenge Sets","date":"2024-01-29","arxiv_id":"2401.16313","repositories_listed":1,"syntology":null},{"url":"/paper/topro-token-level-prompt-decomposition-for","slug":"topro-token-level-prompt-decomposition-for","title":"ToPro: Token-Level Prompt Decomposition for Cross-Lingual Sequence Labeling Tasks","date":"2024-01-29","arxiv_id":"2401.16589","repositories_listed":1,"syntology":null},{"url":"/paper/ppm-automated-generation-of-diverse","slug":"ppm-automated-generation-of-diverse","title":"PPM: Automated Generation of Diverse Programming Problems for Benchmarking Code Generation Models","date":"2024-01-28","arxiv_id":"2401.15545","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ppm-automated-generation-of-diverse#ran","syntology_url":"https://syntology.ai/paper/2401.15545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15545"}},"official":{"repos":["seekingdream/ppm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-the-fairness-of-image-upsampling","slug":"benchmarking-the-fairness-of-image-upsampling","title":"Benchmarking the Fairness of Image Upsampling Methods","date":"2024-01-24","arxiv_id":"2401.13555","repositories_listed":1,"syntology":null},{"url":"/paper/dataset-and-benchmark-novel-sensors-for","slug":"dataset-and-benchmark-novel-sensors-for","title":"Dataset and Benchmark: Novel Sensors for Autonomous Vehicle Perception","date":"2024-01-24","arxiv_id":"2401.13853","repositories_listed":1,"syntology":null},{"url":"/paper/scimmir-benchmarking-scientific-multi-modal","slug":"scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","arxiv_id":"2401.13478","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scimmir-benchmarking-scientific-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2401.13478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13478"}},"official":{"repos":["wusiwei0410/scimmir"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llms-via-uncertainty","slug":"benchmarking-llms-via-uncertainty","title":"Benchmarking LLMs via Uncertainty Quantification","date":"2024-01-23","arxiv_id":"2401.12794","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":1,"n_instrument":12,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 12 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-llms-via-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2401.12794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12794"}},"official":{"repos":["smartyfh/llm-uncertainty-bench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-neural-network-benchmarks-for-selective","slug":"deep-neural-network-benchmarks-for-selective","title":"Deep Neural Network Benchmarks for Selective Classification","date":"2024-01-23","arxiv_id":"2401.12708","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/deep-neural-network-benchmarks-for-selective#ran","syntology_url":"https://syntology.ai/paper/2401.12708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12708"}},"official":{"repos":["andrepugni/esc"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/llpowershap-logistic-loss-based-automated","slug":"llpowershap-logistic-loss-based-automated","title":"LLpowershap: Logistic Loss-based Automated Shapley Values Feature Selection Method","date":"2024-01-23","arxiv_id":"2401.12683","repositories_listed":1,"syntology":null},{"url":"/paper/what-the-weight-a-unified-framework-for-zero","slug":"what-the-weight-a-unified-framework-for-zero","title":"What the Weight?! A Unified Framework for Zero-Shot Knowledge Composition","date":"2024-01-23","arxiv_id":"2401.12756","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-multimodal-models-against","slug":"benchmarking-large-multimodal-models-against","title":"Benchmarking Large Multimodal Models against Common Corruptions","date":"2024-01-22","arxiv_id":"2401.11943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-multimodal-models-against#ran","syntology_url":"https://syntology.ai/paper/2401.11943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11943"}},"official":{"repos":["sail-sg/mmcbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chexagent-towards-a-foundation-model-for","slug":"chexagent-towards-a-foundation-model-for","title":"A Vision-Language Foundation Model to Enhance Efficiency of Chest X-ray Interpretation","date":"2024-01-22","arxiv_id":"2401.12208","repositories_listed":1,"syntology":null},{"url":"/paper/subgroup-analysis-methods-for-time-to-event","slug":"subgroup-analysis-methods-for-time-to-event","title":"Subgroup analysis methods for time-to-event outcomes in heterogeneous randomized controlled trials","date":"2024-01-22","arxiv_id":"2401.11842","repositories_listed":1,"syntology":null},{"url":"/paper/r-judge-benchmarking-safety-risk-awareness","slug":"r-judge-benchmarking-safety-risk-awareness","title":"R-Judge: Benchmarking Safety Risk Awareness for LLM Agents","date":"2024-01-18","arxiv_id":"2401.10019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r-judge-benchmarking-safety-risk-awareness#ran","syntology_url":"https://syntology.ai/paper/2401.10019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10019"}},"official":{"repos":["lordog/r-judge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-the-robustness-of-image","slug":"benchmarking-the-robustness-of-image","title":"WAVES: Benchmarking the Robustness of Image Watermarks","date":"2024-01-16","arxiv_id":"2401.08573","repositories_listed":1,"syntology":{"n":21,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":21,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/benchmarking-the-robustness-of-image#ran","syntology_url":"https://syntology.ai/paper/2401.08573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08573"}},"official":{"repos":["umd-huang-lab/waves"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/harnessing-orthogonality-to-train-low-rank","slug":"harnessing-orthogonality-to-train-low-rank","title":"Harnessing Orthogonality to Train Low-Rank Neural Networks","date":"2024-01-16","arxiv_id":"2401.08505","repositories_listed":1,"syntology":null},{"url":"/paper/opendpd-an-open-source-end-to-end-learning","slug":"opendpd-an-open-source-end-to-end-learning","title":"OpenDPD: An Open-Source End-to-End Learning & Benchmarking Framework for Wideband Power Amplifier Modeling and Digital Pre-Distortion","date":"2024-01-16","arxiv_id":"2401.08318","repositories_listed":1,"syntology":null},{"url":"/paper/rsud20k-a-dataset-for-road-scene","slug":"rsud20k-a-dataset-for-road-scene","title":"RSUD20K: A Dataset for Road Scene Understanding In Autonomous Driving","date":"2024-01-14","arxiv_id":"2401.07322","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rsud20k-a-dataset-for-road-scene#ran","syntology_url":"https://syntology.ai/paper/2401.07322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07322"}},"official":{"repos":["hasibzunair/rsud20k"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/infiagent-dabench-evaluating-agents-on-data","slug":"infiagent-dabench-evaluating-agents-on-data","title":"InfiAgent-DABench: Evaluating Agents on Data Analysis Tasks","date":"2024-01-10","arxiv_id":"2401.05507","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/infiagent-dabench-evaluating-agents-on-data#ran","syntology_url":"https://syntology.ai/paper/2401.05507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05507"}},"official":{"repos":["infiagent/infiagent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepspeed-fastgen-high-throughput-text","slug":"deepspeed-fastgen-high-throughput-text","title":"DeepSpeed-FastGen: High-throughput Text Generation for LLMs via MII and DeepSpeed-Inference","date":"2024-01-09","arxiv_id":"2401.08671","repositories_listed":1,"syntology":null},{"url":"/paper/mst-adaptive-multi-scale-tokens-guided","slug":"mst-adaptive-multi-scale-tokens-guided","title":"MST: Adaptive Multi-Scale Tokens Guided Interactive Segmentation","date":"2024-01-09","arxiv_id":"2401.04403","repositories_listed":1,"syntology":null},{"url":"/paper/global-prediction-of-covid-19-variant","slug":"global-prediction-of-covid-19-variant","title":"Global Prediction of COVID-19 Variant Emergence Using Dynamics-Informed Graph Neural Networks","date":"2024-01-07","arxiv_id":"2401.03390","repositories_listed":1,"syntology":null},{"url":"/paper/segment-anything-model-for-medical-image-1","slug":"segment-anything-model-for-medical-image-1","title":"Segment Anything Model for Medical Image Segmentation: Current Applications and Future Directions","date":"2024-01-07","arxiv_id":"2401.03495","repositories_listed":1,"syntology":null},{"url":"/paper/caviar-co-simulation-of-6g-communications-3d","slug":"caviar-co-simulation-of-6g-communications-3d","title":"CAVIAR: Co-simulation of 6G Communications, 3D Scenarios and AI for Digital Twins","date":"2024-01-06","arxiv_id":"2401.03310","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-3d-air-signature-by-pen-tip-tail","slug":"enhancing-3d-air-signature-by-pen-tip-tail","title":"Enhancing 3D-Air Signature by Pen Tip Tail Trajectory Awareness: Dataset and Featuring by Novel Spatio-temporal CNN","date":"2024-01-05","arxiv_id":"2401.02649","repositories_listed":1,"syntology":null},{"url":"/paper/german-text-embedding-clustering-benchmark","slug":"german-text-embedding-clustering-benchmark","title":"German Text Embedding Clustering Benchmark","date":"2024-01-05","arxiv_id":"2401.02709","repositories_listed":1,"syntology":null},{"url":"/paper/a-call-to-reflect-on-evaluation-practices-for-1","slug":"a-call-to-reflect-on-evaluation-practices-for-1","title":"A Call to Reflect on Evaluation Practices for Age Estimation: Comparative Analysis of the State-of-the-Art and a Unified Benchmark","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/am-radio-agglomerative-vision-foundation","slug":"am-radio-agglomerative-vision-foundation","title":"AM-RADIO: Agglomerative Vision Foundation Model Reduce All Domains Into One","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-on","slug":"benchmarking-large-language-models-on","title":"Benchmarking Large Language Models on Controllable Generation under Diversified Instructions","date":"2024-01-01","arxiv_id":"2401.00690","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2401.00690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00690"}},"official":{"repos":["xt-cyh/codi-eval"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/bibench-benchmarking-data-analysis-knowledge","slug":"bibench-benchmarking-data-analysis-knowledge","title":"FinDABench: Benchmarking Financial Data Analysis Ability of Large Language Models","date":"2024-01-01","arxiv_id":"2401.02982","repositories_listed":1,"syntology":null},{"url":"/paper/seed-bench-benchmarking-multimodal-large","slug":"seed-bench-benchmarking-multimodal-large","title":"SEED-Bench: Benchmarking Multimodal Large Language Models","date":"2024-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/tspp-a-unified-benchmarking-tool-for-time","slug":"tspp-a-unified-benchmarking-tool-for-time","title":"TSPP: A Unified Benchmarking Tool for Time-series Forecasting","date":"2023-12-28","arxiv_id":"2312.17100","repositories_listed":1,"syntology":null},{"url":"/paper/aptv2-benchmarking-animal-pose-estimation-and","slug":"aptv2-benchmarking-animal-pose-estimation-and","title":"APTv2: Benchmarking Animal Pose Estimation and Tracking with a Large-scale Dataset and Beyond","date":"2023-12-25","arxiv_id":"2312.15612","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-defending-against-indirect","slug":"benchmarking-and-defending-against-indirect","title":"Benchmarking and Defending Against Indirect Prompt Injection Attacks on Large Language Models","date":"2023-12-21","arxiv_id":"2312.14197","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-and-defending-against-indirect#ran","syntology_url":"https://syntology.ai/paper/2312.14197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14197"}},"official":{"repos":["microsoft/BIPIA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/retailsynth-synthetic-data-generation-for","slug":"retailsynth-synthetic-data-generation-for","title":"RetailSynth: Synthetic Data Generation for Retail AI Systems Evaluation","date":"2023-12-21","arxiv_id":"2312.14095","repositories_listed":1,"syntology":null},{"url":"/paper/comparing-machine-learning-algorithms-by","slug":"comparing-machine-learning-algorithms-by","title":"Comparing Machine Learning Algorithms by Union-Free Generic Depth","date":"2023-12-20","arxiv_id":"2312.12839","repositories_listed":1,"syntology":null},{"url":"/paper/fifar-a-fraud-detection-dataset-for-learning","slug":"fifar-a-fraud-detection-dataset-for-learning","title":"FiFAR: A Fraud Detection Dataset for Learning to Defer","date":"2023-12-20","arxiv_id":"2312.13218","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fifar-a-fraud-detection-dataset-for-learning#ran","syntology_url":"https://syntology.ai/paper/2312.13218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13218"}},"official":{"repos":["feedzai/fifar-dataset"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-compute-is-not-all-you-need-for","slug":"scaling-compute-is-not-all-you-need-for","title":"Scaling Compute Is Not All You Need for Adversarial Robustness","date":"2023-12-20","arxiv_id":"2312.13131","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/scaling-compute-is-not-all-you-need-for#ran","syntology_url":"https://syntology.ai/paper/2312.13131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13131"}},"official":{"repos":["dedeswim/timm-adv-training"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tracking-any-object-amodally","slug":"tracking-any-object-amodally","title":"TAO-Amodal: A Benchmark for Tracking Any Object Amodally","date":"2023-12-19","arxiv_id":"2312.12433","repositories_listed":1,"syntology":null},{"url":"/paper/code-ownership-in-open-source-ai-software","slug":"code-ownership-in-open-source-ai-software","title":"Code Ownership in Open-Source AI Software Security","date":"2023-12-18","arxiv_id":"2312.10861","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-train-neural-field-representations-a","slug":"how-to-train-neural-field-representations-a","title":"How to Train Neural Field Representations: A Comprehensive Study and Benchmark","date":"2023-12-16","arxiv_id":"2312.10531","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-train-neural-field-representations-a#ran","syntology_url":"https://syntology.ai/paper/2312.10531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10531"}},"official":{"repos":["samuelepapa/fit-a-nef"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/binary-code-summarization-benchmarking","slug":"binary-code-summarization-benchmarking","title":"Binary Code Summarization: Benchmarking ChatGPT/GPT-4 and Other Large Language Models","date":"2023-12-15","arxiv_id":"2312.09601","repositories_listed":1,"syntology":null},{"url":"/paper/how-well-does-gpt-4v-ision-adapt-to","slug":"how-well-does-gpt-4v-ision-adapt-to","title":"How Well Does GPT-4V(ision) Adapt to Distribution Shifts? A Preliminary Investigation","date":"2023-12-12","arxiv_id":"2312.07424","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-well-does-gpt-4v-ision-adapt-to#ran","syntology_url":"https://syntology.ai/paper/2312.07424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07424"}},"official":{"repos":["jameszhou-gl/gpt-4v-distribution-shift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-survey-on-outlier-and-anomaly-detection","slug":"meta-survey-on-outlier-and-anomaly-detection","title":"Meta-survey on outlier and anomaly detection","date":"2023-12-12","arxiv_id":"2312.07101","repositories_listed":1,"syntology":null},{"url":"/paper/egoplan-bench-benchmarking-egocentric","slug":"egoplan-bench-benchmarking-egocentric","title":"EgoPlan-Bench: Benchmarking Multimodal Large Language Models for Human-Level Planning","date":"2023-12-11","arxiv_id":"2312.06722","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/egoplan-bench-benchmarking-egocentric#ran","syntology_url":"https://syntology.ai/paper/2312.06722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06722"}},"official":{"repos":["chenyi99/egoplan"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/eq-bench-an-emotional-intelligence-benchmark","slug":"eq-bench-an-emotional-intelligence-benchmark","title":"EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models","date":"2023-12-11","arxiv_id":"2312.06281","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eq-bench-an-emotional-intelligence-benchmark#ran","syntology_url":"https://syntology.ai/paper/2312.06281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06281"}},"official":{"repos":["eq-bench/eq-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-distribution-shift-in-tabular-1","slug":"benchmarking-distribution-shift-in-tabular-1","title":"Benchmarking Distribution Shift in Tabular Data with TableShift","date":"2023-12-10","arxiv_id":"2312.07577","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-distribution-shift-in-tabular-1#ran","syntology_url":"https://syntology.ai/paper/2312.07577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07577"}},"official":{"repos":["mlfoundations/tableshift"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-of-query-strategies-towards","slug":"benchmarking-of-query-strategies-towards","title":"Benchmarking of Query Strategies: Towards Future Deep Active Learning","date":"2023-12-10","arxiv_id":"2312.05751","repositories_listed":1,"syntology":null},{"url":"/paper/streamline-an-automated-machine-learning","slug":"streamline-an-automated-machine-learning","title":"STREAMLINE: An Automated Machine Learning Pipeline for Biomedicine Applied to Examine the Utility of Photography-Based Phenotypes for OSA Prediction Across International Sleep Centers","date":"2023-12-09","arxiv_id":"2312.05461","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-and-analysis-of-unsupervised","slug":"benchmarking-and-analysis-of-unsupervised","title":"Benchmarking and Analysis of Unsupervised Object Segmentation from Real-world Single Images","date":"2023-12-08","arxiv_id":"2312.04947","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-analysis-of-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2312.04947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04947"}},"official":{"repos":["vlar-group/unsupobjseg"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-language-agents-be-alternatives-to-ppo-a","slug":"can-language-agents-be-alternatives-to-ppo-a","title":"Can language agents be alternatives to PPO? A Preliminary Empirical Study On OpenAI Gym","date":"2023-12-06","arxiv_id":"2312.03290","repositories_listed":1,"syntology":null},{"url":"/paper/dyport-dynamic-importance-based-hypothesis","slug":"dyport-dynamic-importance-based-hypothesis","title":"Dyport: Dynamic Importance-based Hypothesis Generation Benchmarking Technique","date":"2023-12-06","arxiv_id":"2312.03303","repositories_listed":1,"syntology":null},{"url":"/paper/khabarchin-automatic-detection-of-important","slug":"khabarchin-automatic-detection-of-important","title":"KhabarChin: Automatic Detection of Important News in the Persian Language","date":"2023-12-06","arxiv_id":"2312.03361","repositories_listed":1,"syntology":null},{"url":"/paper/pearl-a-production-ready-reinforcement","slug":"pearl-a-production-ready-reinforcement","title":"Pearl: A Production-ready Reinforcement Learning Agent","date":"2023-12-06","arxiv_id":"2312.03814","repositories_listed":1,"syntology":null},{"url":"/paper/bedd-the-minerl-basalt-evaluation-and-1","slug":"bedd-the-minerl-basalt-evaluation-and-1","title":"BEDD: The MineRL BASALT Evaluation and Demonstrations Dataset for Training and Benchmarking Agents that Solve Fuzzy Tasks","date":"2023-12-05","arxiv_id":"2312.02405","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bedd-the-minerl-basalt-evaluation-and-1#ran","syntology_url":"https://syntology.ai/paper/2312.02405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02405"}},"official":{"repos":["minerllabs/basalt-benchmark"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/let-the-llms-talk-simulating-human-to-human","slug":"let-the-llms-talk-simulating-human-to-human","title":"Let the LLMs Talk: Simulating Human-to-Human Conversational QA via Zero-Shot LLM-to-LLM Interactions","date":"2023-12-05","arxiv_id":"2312.02913","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarl-benchmarking-multi-agent","slug":"benchmarl-benchmarking-multi-agent","title":"BenchMARL: Benchmarking Multi-Agent Reinforcement Learning","date":"2023-12-03","arxiv_id":"2312.01472","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarl-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2312.01472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01472"}},"official":{"repos":["facebookresearch/benchmarl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alignbench-benchmarking-chinese-alignment-of","slug":"alignbench-benchmarking-chinese-alignment-of","title":"AlignBench: Benchmarking Chinese Alignment of Large Language Models","date":"2023-11-30","arxiv_id":"2311.18743","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alignbench-benchmarking-chinese-alignment-of#ran","syntology_url":"https://syntology.ai/paper/2311.18743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18743"}},"official":{"repos":["thudm/alignbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controlgym-large-scale-safety-critical","slug":"controlgym-large-scale-safety-critical","title":"Controlgym: Large-Scale Control Environments for Benchmarking Reinforcement Learning Algorithms","date":"2023-11-30","arxiv_id":"2311.18736","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-ligand-pose-sampling-for-molecular","slug":"enhancing-ligand-pose-sampling-for-molecular","title":"Enhancing Ligand Pose Sampling for Molecular Docking","date":"2023-11-30","arxiv_id":"2312.00191","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-ligand-pose-sampling-for-molecular#ran","syntology_url":"https://syntology.ai/paper/2312.00191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00191"}},"official":{"repos":["drorlab/glow_ives"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mathbb-z-2-times-mathbb-z-2-equivariant","slug":"mathbb-z-2-times-mathbb-z-2-equivariant","title":"$\\mathbb{Z}_2\\times \\mathbb{Z}_2$ Equivariant Quantum Neural Networks: Benchmarking against Classical Neural Networks","date":"2023-11-30","arxiv_id":"2311.18744","repositories_listed":1,"syntology":null},{"url":"/paper/taskbench-benchmarking-large-language-models","slug":"taskbench-benchmarking-large-language-models","title":"TaskBench: Benchmarking Large Language Models for Task Automation","date":"2023-11-30","arxiv_id":"2311.18760","repositories_listed":1,"syntology":null},{"url":"/paper/towards-assessing-and-benchmarking-risk","slug":"towards-assessing-and-benchmarking-risk","title":"Towards Assessing and Benchmarking Risk-Return Tradeoff of Off-Policy Evaluation","date":"2023-11-30","arxiv_id":"2311.18207","repositories_listed":1,"syntology":null},{"url":"/paper/are-we-going-mad-benchmarking-multi-agent","slug":"are-we-going-mad-benchmarking-multi-agent","title":"Should we be going MAD? A Look at Multi-Agent Debate Strategies for LLMs","date":"2023-11-29","arxiv_id":"2311.17371","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-we-going-mad-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2311.17371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17371"}},"official":{"repos":["instadeepai/debatellm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedical-knowledge-graph-enhanced-prompt","slug":"biomedical-knowledge-graph-enhanced-prompt","title":"Biomedical knowledge graph-optimized prompt generation for large language models","date":"2023-11-29","arxiv_id":"2311.17330","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/biomedical-knowledge-graph-enhanced-prompt#ran","syntology_url":"https://syntology.ai/paper/2311.17330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17330"}},"official":{"repos":["BaranziniLab/KG_RAG"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/experimental-analysis-of-large-scale","slug":"experimental-analysis-of-large-scale","title":"Experimental Analysis of Large-scale Learnable Vector Storage Compression","date":"2023-11-27","arxiv_id":"2311.15578","repositories_listed":1,"syntology":null},{"url":"/paper/uhgeval-benchmarking-the-hallucination-of","slug":"uhgeval-benchmarking-the-hallucination-of","title":"UHGEval: Benchmarking the Hallucination of Chinese Large Language Models via Unconstrained Generation","date":"2023-11-26","arxiv_id":"2311.15296","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uhgeval-benchmarking-the-hallucination-of#ran","syntology_url":"https://syntology.ai/paper/2311.15296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15296"}},"official":{"repos":["IAAR-Shanghai/UHGEval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-robustness-of-text-image","slug":"benchmarking-robustness-of-text-image","title":"Benchmarking Robustness of Text-Image Composed Retrieval","date":"2023-11-24","arxiv_id":"2311.14837","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-robustness-of-text-image#ran","syntology_url":"https://syntology.ai/paper/2311.14837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.14837"}},"official":{"repos":["suntongtongtong/benchmark-robustness-text-image-compose-retrieval"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/creating-and-benchmarking-a-synthetic-dataset","slug":"creating-and-benchmarking-a-synthetic-dataset","title":"Creating and Leveraging a Synthetic Dataset of Cloud Optical Thickness Measures for Cloud Detection in MSI","date":"2023-11-23","arxiv_id":"2311.14024","repositories_listed":1,"syntology":null},{"url":"/paper/dialogue-quality-and-emotion-annotations-for","slug":"dialogue-quality-and-emotion-annotations-for","title":"Dialogue Quality and Emotion Annotations for Customer Support Conversations","date":"2023-11-23","arxiv_id":"2311.13910","repositories_listed":1,"syntology":null},{"url":"/paper/learning-dynamic-selection-and-pricing-of-out","slug":"learning-dynamic-selection-and-pricing-of-out","title":"Learning Dynamic Selection and Pricing of Out-of-Home Deliveries","date":"2023-11-23","arxiv_id":"2311.13983","repositories_listed":1,"syntology":null},{"url":"/paper/a-projected-nonlinear-state-space-model-for","slug":"a-projected-nonlinear-state-space-model-for","title":"A projected nonlinear state-space model for forecasting time series signals","date":"2023-11-22","arxiv_id":"2311.13247","repositories_listed":1,"syntology":null},{"url":"/paper/pg-video-llava-pixel-grounding-large-video","slug":"pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","arxiv_id":"2311.13435","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pg-video-llava-pixel-grounding-large-video#ran","syntology_url":"https://syntology.ai/paper/2311.13435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13435"}},"official":{"repos":["mbzuai-oryx/video-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bend-benchmarking-dna-language-models-on","slug":"bend-benchmarking-dna-language-models-on","title":"BEND: Benchmarking DNA Language Models on biologically meaningful tasks","date":"2023-11-21","arxiv_id":"2311.12570","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/bend-benchmarking-dna-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2311.12570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.12570"}},"official":{"repos":["frederikkemarin/bend"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/imgtb-a-framework-for-machine-generated-text","slug":"imgtb-a-framework-for-machine-generated-text","title":"IMGTB: A Framework for Machine-Generated Text Detection Benchmarking","date":"2023-11-21","arxiv_id":"2311.12574","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-more-inductive-world-for-drug","slug":"towards-a-more-inductive-world-for-drug","title":"Towards a more inductive world for drug repurposing approaches","date":"2023-11-21","arxiv_id":"2311.12670","repositories_listed":1,"syntology":null},{"url":"/paper/a-good-feature-extractor-is-all-you-need-for","slug":"a-good-feature-extractor-is-all-you-need-for","title":"Benchmarking Pathology Feature Extractors for Whole Slide Image Classification","date":"2023-11-20","arxiv_id":"2311.11772","repositories_listed":1,"syntology":null}],"record_sha256":"6f47695ef09f65ad5e99ffb26c2a366a13b4eb4395558a9770088f8569169451","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}