{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/ran/4","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":4,"pages_in_order":8,"rows_per_page":100,"rows":[301,400],"of":749,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking/papers/ran/1","prev":"/task/benchmarking/papers/ran/3","next":"/task/benchmarking/papers/ran/5","papers":[{"url":"/paper/a-review-and-efficient-implementation-of","slug":"a-review-and-efficient-implementation-of","title":"A Review and Efficient Implementation of Scene Graph Generation Metrics","date":"2024-04-15","arxiv_id":"2404.09616","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-review-and-efficient-implementation-of#ran","syntology_url":"https://syntology.ai/paper/2404.09616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09616"}},"official":{"repos":["lorjul/sgbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llama2-mistral-gemma-and-gpt-for","slug":"benchmarking-llama2-mistral-gemma-and-gpt-for","title":"Benchmarking Llama2, Mistral, Gemma and GPT for Factuality, Toxicity, Bias and Propensity for Hallucinations","date":"2024-04-15","arxiv_id":"2404.09785","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-llama2-mistral-gemma-and-gpt-for#ran","syntology_url":"https://syntology.ai/paper/2404.09785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09785"}},"official":{"repos":["innodatalabs/innodata-llm-safety"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/osworld-benchmarking-multimodal-agents-for","slug":"osworld-benchmarking-multimodal-agents-for","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","date":"2024-04-11","arxiv_id":"2404.07972","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/osworld-benchmarking-multimodal-agents-for#ran","syntology_url":"https://syntology.ai/paper/2404.07972","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07972"}},"official":{"repos":["xlang-ai/OSWorld"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-your-llm-outdated-benchmarking-llms","slug":"is-your-llm-outdated-benchmarking-llms","title":"DyKnow: Dynamically Verifying Time-Sensitive Factual Knowledge in LLMs","date":"2024-04-10","arxiv_id":"2404.08700","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-your-llm-outdated-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2404.08700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08700"}},"official":{"repos":["sislab-unitn/dyknow"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentquest-a-modular-benchmark-framework-to","slug":"agentquest-a-modular-benchmark-framework-to","title":"AgentQuest: A Modular Benchmark Framework to Measure Progress and Improve LLM Agents","date":"2024-04-09","arxiv_id":"2404.06411","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/agentquest-a-modular-benchmark-framework-to#ran","syntology_url":"https://syntology.ai/paper/2404.06411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.06411"}},"official":{"repos":["nec-research/agentquest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pollmgraph-unraveling-hallucinations-in-large","slug":"pollmgraph-unraveling-hallucinations-in-large","title":"PoLLMgraph: Unraveling Hallucinations in Large Language Models via State Transition Dynamics","date":"2024-04-06","arxiv_id":"2404.04722","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pollmgraph-unraveling-hallucinations-in-large#ran","syntology_url":"https://syntology.ai/paper/2404.04722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04722"}},"official":{"repos":["hitum-dev/pollmgraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-improving-compositional","slug":"benchmarking-and-improving-compositional","title":"Benchmarking and Improving Compositional Generalization of Multi-aspect Controllable Text Generation","date":"2024-04-05","arxiv_id":"2404.04232","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-improving-compositional#ran","syntology_url":"https://syntology.ai/paper/2404.04232","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04232"}},"official":{"repos":["tqzhong/cg4mctg"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/who-evaluates-the-evaluations-objectively","slug":"who-evaluates-the-evaluations-objectively","title":"Who Evaluates the Evaluations? Objectively Scoring Text-to-Image Prompt Coherence Metrics with T2IScoreScore (TS2)","date":"2024-04-05","arxiv_id":"2404.04251","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/who-evaluates-the-evaluations-objectively#ran","syntology_url":"https://syntology.ai/paper/2404.04251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04251"}},"official":{"repos":["michaelsaxon/T2IScoreScore"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-chatgpt-on-algorithmic-reasoning","slug":"benchmarking-chatgpt-on-algorithmic-reasoning","title":"Benchmarking ChatGPT on Algorithmic Reasoning","date":"2024-04-04","arxiv_id":"2404.03441","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-chatgpt-on-algorithmic-reasoning#ran","syntology_url":"https://syntology.ai/paper/2404.03441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03441"}},"official":{"repos":["mcleish7/clrs4lm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/outlier-efficient-hopfield-layers-for-large","slug":"outlier-efficient-hopfield-layers-for-large","title":"Outlier-Efficient Hopfield Layers for Large Transformer-Based Models","date":"2024-04-04","arxiv_id":"2404.03828","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":5,"n_ran_checked":5,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/outlier-efficient-hopfield-layers-for-large#ran","syntology_url":"https://syntology.ai/paper/2404.03828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03828"}},"official":{"repos":["magics-lab/outeffhop"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/no-zero-shot-without-exponential-data","slug":"no-zero-shot-without-exponential-data","title":"No \"Zero-Shot\" Without Exponential Data: Pretraining Concept Frequency Determines Multimodal Model Performance","date":"2024-04-04","arxiv_id":"2404.04125","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/no-zero-shot-without-exponential-data#ran","syntology_url":"https://syntology.ai/paper/2404.04125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04125"}},"official":{"repos":["bethgelab/frequency_determines_performance"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/atom-level-optical-chemical-structure","slug":"atom-level-optical-chemical-structure","title":"Atom-Level Optical Chemical Structure Recognition with Limited Supervision","date":"2024-04-02","arxiv_id":"2404.01743","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/atom-level-optical-chemical-structure#ran","syntology_url":"https://syntology.ai/paper/2404.01743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01743"}},"official":{"repos":["molden/atomlenz"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prego-online-mistake-detection-in-procedural","slug":"prego-online-mistake-detection-in-procedural","title":"PREGO: online mistake detection in PRocedural EGOcentric videos","date":"2024-04-02","arxiv_id":"2404.01933","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prego-online-mistake-detection-in-procedural#ran","syntology_url":"https://syntology.ai/paper/2404.01933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01933"}},"official":{"repos":["aleflabo/prego"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-llm-reasoning-generalists-with","slug":"advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","arxiv_id":"2404.02078","repositories_listed":1,"syntology":{"n":20,"n_ran":18,"n_constructed":0,"n_ran_checked":13,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/advancing-llm-reasoning-generalists-with#ran","syntology_url":"https://syntology.ai/paper/2404.02078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02078"}},"official":{"repos":["openbmb/eurus"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tfb-towards-comprehensive-and-fair","slug":"tfb-towards-comprehensive-and-fair","title":"TFB: Towards Comprehensive and Fair Benchmarking of Time Series Forecasting Methods","date":"2024-03-29","arxiv_id":"2403.20150","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tfb-towards-comprehensive-and-fair#ran","syntology_url":"https://syntology.ai/paper/2403.20150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20150"}},"official":{"repos":["decisionintelligence/tfb"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-the-robustness-of-temporal","slug":"benchmarking-the-robustness-of-temporal","title":"Benchmarking the Robustness of Temporal Action Detection Models Against Temporal Corruptions","date":"2024-03-29","arxiv_id":"2403.20254","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-the-robustness-of-temporal#ran","syntology_url":"https://syntology.ai/paper/2403.20254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20254"}},"official":{"repos":["alvin-zeng/temporal-robustness-benchmark"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-counterfactual-image-generation","slug":"benchmarking-counterfactual-image-generation","title":"Benchmarking Counterfactual Image Generation","date":"2024-03-29","arxiv_id":"2403.20287","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-counterfactual-image-generation#ran","syntology_url":"https://syntology.ai/paper/2403.20287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20287"}},"official":{"repos":["gulnazaki/counterfactual-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-good-at-utility","slug":"are-large-language-models-good-at-utility","title":"Are Large Language Models Good at Utility Judgments?","date":"2024-03-28","arxiv_id":"2403.19216","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/are-large-language-models-good-at-utility#ran","syntology_url":"https://syntology.ai/paper/2403.19216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19216"}},"official":{"repos":["ict-bigdatalab/utility_judgments"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/imagenet-d-benchmarking-neural-network","slug":"imagenet-d-benchmarking-neural-network","title":"ImageNet-D: Benchmarking Neural Network Robustness on Diffusion Synthetic Object","date":"2024-03-27","arxiv_id":"2403.18775","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imagenet-d-benchmarking-neural-network#ran","syntology_url":"https://syntology.ai/paper/2403.18775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18775"}},"official":{"repos":["chenshuang-zhang/imagenet_d"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-object-detectors-with-coco-a-new","slug":"benchmarking-object-detectors-with-coco-a-new","title":"Benchmarking Object Detectors with COCO: A New Path Forward","date":"2024-03-27","arxiv_id":"2403.18819","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-object-detectors-with-coco-a-new#ran","syntology_url":"https://syntology.ai/paper/2403.18819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18819"}},"official":{"repos":["kdexd/coco-rem"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic","slug":"arabicaqa-a-comprehensive-dataset-for-arabic","title":"ArabicaQA: A Comprehensive Dataset for Arabic Question Answering","date":"2024-03-26","arxiv_id":"2403.17848","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arabicaqa-a-comprehensive-dataset-for-arabic#ran","syntology_url":"https://syntology.ai/paper/2403.17848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17848"}},"official":{"repos":["datascienceuibk/arabicaqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-fragility-of-active-learners","slug":"on-the-fragility-of-active-learners","title":"On the Fragility of Active Learners for Text Classification","date":"2024-03-23","arxiv_id":"2403.15744","repositories_listed":1,"syntology":{"n":15,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/on-the-fragility-of-active-learners#ran","syntology_url":"https://syntology.ai/paper/2403.15744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15744"}},"official":{"repos":["ThuongTNguyen/ALchemist"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-chinese-commonsense-reasoning-of","slug":"benchmarking-chinese-commonsense-reasoning-of","title":"Benchmarking Chinese Commonsense Reasoning of LLMs: From Chinese-Specifics to Reasoning-Memorization Correlations","date":"2024-03-21","arxiv_id":"2403.14112","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-chinese-commonsense-reasoning-of#ran","syntology_url":"https://syntology.ai/paper/2403.14112","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14112"}},"official":{"repos":["opendatalab/charm"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/real-iad-a-real-world-multi-view-dataset-for","slug":"real-iad-a-real-world-multi-view-dataset-for","title":"Real-IAD: A Real-World Multi-View Dataset for Benchmarking Versatile Industrial Anomaly Detection","date":"2024-03-19","arxiv_id":"2403.12580","repositories_listed":2,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/real-iad-a-real-world-multi-view-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2403.12580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12580"}},"official":null}},{"url":"/paper/alphafin-benchmarking-financial-analysis-with","slug":"alphafin-benchmarking-financial-analysis-with","title":"AlphaFin: Benchmarking Financial Analysis with Retrieval-Augmented Stock-Chain Framework","date":"2024-03-19","arxiv_id":"2403.12582","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/alphafin-benchmarking-financial-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2403.12582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12582"}},"official":{"repos":["alphafin-proj/alphafin"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/melting-point-mobile-evaluation-of-language","slug":"melting-point-mobile-evaluation-of-language","title":"MELTing point: Mobile Evaluation of Language Transformers","date":"2024-03-19","arxiv_id":"2403.12844","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/melting-point-mobile-evaluation-of-language#ran","syntology_url":"https://syntology.ai/paper/2403.12844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12844"}},"official":{"repos":["brave-experiments/melt-public"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/an-improved-metric-and-benchmark-for","slug":"an-improved-metric-and-benchmark-for","title":"An Improved Metric and Benchmark for Assessing the Performance of Virtual Screening Models","date":"2024-03-15","arxiv_id":"2403.10478","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-improved-metric-and-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2403.10478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.10478"}},"official":{"repos":["molecularmodelinglab/bigbind"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-drafter-for-fast-speculative","slug":"recurrent-drafter-for-fast-speculative","title":"Recurrent Drafter for Fast Speculative Decoding in Large Language Models","date":"2024-03-14","arxiv_id":"2403.09919","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/recurrent-drafter-for-fast-speculative#ran","syntology_url":"https://syntology.ai/paper/2403.09919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.09919"}},"official":{"repos":["apple/ml-recurrent-drafter"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stabletoolbench-towards-stable-large-scale","slug":"stabletoolbench-towards-stable-large-scale","title":"StableToolBench: Towards Stable Large-Scale Benchmarking on Tool Learning of Large Language Models","date":"2024-03-12","arxiv_id":"2403.07714","repositories_listed":4,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":5,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/stabletoolbench-towards-stable-large-scale#ran","syntology_url":"https://syntology.ai/paper/2403.07714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07714"}},"official":{"repos":["thunlp-mt/stabletoolbench","zhichengg/stabletoolbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/better-than-classical-the-subtle-art-of","slug":"better-than-classical-the-subtle-art-of","title":"Better than classical? The subtle art of benchmarking quantum machine learning models","date":"2024-03-11","arxiv_id":"2403.07059","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-than-classical-the-subtle-art-of#ran","syntology_url":"https://syntology.ai/paper/2403.07059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07059"}},"official":{"repos":["xanaduai/qml-benchmarks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/benchmarking-micro-action-recognition-dataset","slug":"benchmarking-micro-action-recognition-dataset","title":"Benchmarking Micro-action Recognition: Dataset, Methods, and Applications","date":"2024-03-08","arxiv_id":"2403.05234","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-micro-action-recognition-dataset#ran","syntology_url":"https://syntology.ai/paper/2403.05234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05234"}},"official":{"repos":["vut-hfut/micro-action"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tapilot-crossing-benchmarking-and-evolving","slug":"tapilot-crossing-benchmarking-and-evolving","title":"Tapilot-Crossing: Benchmarking and Evolving LLMs Towards Interactive Data Analysis Agents","date":"2024-03-08","arxiv_id":"2403.05307","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tapilot-crossing-benchmarking-and-evolving#ran","syntology_url":"https://syntology.ai/paper/2403.05307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05307"}},"official":{"repos":["tapilot-crossing/tapilot_code"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dissecting-sample-hardness-a-fine-grained","slug":"dissecting-sample-hardness-a-fine-grained","title":"Dissecting Sample Hardness: A Fine-Grained Analysis of Hardness Characterization Methods for Data-Centric AI","date":"2024-03-07","arxiv_id":"2403.04551","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dissecting-sample-hardness-a-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2403.04551","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04551"}},"official":{"repos":["seedatnabeel/h-cat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bait-benchmarking-embedding-architectures-for","slug":"bait-benchmarking-embedding-architectures-for","title":"BAIT: Benchmarking (Embedding) Architectures for Interactive Theorem-Proving","date":"2024-03-06","arxiv_id":"2403.03401","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bait-benchmarking-embedding-architectures-for#ran","syntology_url":"https://syntology.ai/paper/2403.03401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03401"}},"official":null}},{"url":"/paper/injecagent-benchmarking-indirect-prompt","slug":"injecagent-benchmarking-indirect-prompt","title":"InjecAgent: Benchmarking Indirect Prompt Injections in Tool-Integrated Large Language Model Agents","date":"2024-03-05","arxiv_id":"2403.02691","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/injecagent-benchmarking-indirect-prompt#ran","syntology_url":"https://syntology.ai/paper/2403.02691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02691"}},"official":{"repos":["uiuc-kang-lab/injecagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sciassess-benchmarking-llm-proficiency-in","slug":"sciassess-benchmarking-llm-proficiency-in","title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","date":"2024-03-04","arxiv_id":"2403.01976","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sciassess-benchmarking-llm-proficiency-in#ran","syntology_url":"https://syntology.ai/paper/2403.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01976"}},"official":{"repos":["sci-assess/sciassess"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/private-benchmarking-to-prevent-contamination","slug":"private-benchmarking-to-prevent-contamination","title":"TRUCE: Private Benchmarking to Prevent Contamination and Improve Comparative Evaluation of LLMs","date":"2024-03-01","arxiv_id":"2403.00393","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/private-benchmarking-to-prevent-contamination#ran","syntology_url":"https://syntology.ai/paper/2403.00393","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00393"}},"official":{"repos":["microsoft/private-benchmarking"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-datasets-a-toolkit-for","slug":"imitation-learning-datasets-a-toolkit-for","title":"Imitation Learning Datasets: A Toolkit For Creating Datasets, Training Agents and Benchmarking","date":"2024-03-01","arxiv_id":"2403.00550","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imitation-learning-datasets-a-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/2403.00550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00550"}},"official":{"repos":["nathangavenski/il-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-uncertainty-disentanglement","slug":"benchmarking-uncertainty-disentanglement","title":"Benchmarking Uncertainty Disentanglement: Specialized Uncertainties for Specialized Tasks","date":"2024-02-29","arxiv_id":"2402.19460","repositories_listed":2,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-uncertainty-disentanglement#ran","syntology_url":"https://syntology.ai/paper/2402.19460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19460"}},"official":{"repos":["bmucsanyi/bud","bmucsanyi/untangle"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/lifelong-benchmarks-efficient-model","slug":"lifelong-benchmarks-efficient-model","title":"Efficient Lifelong Model Evaluation in an Era of Rapid Progress","date":"2024-02-29","arxiv_id":"2402.19472","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/lifelong-benchmarks-efficient-model#ran","syntology_url":"https://syntology.ai/paper/2402.19472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19472"}},"official":{"repos":["bethgelab/sort-and-search"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-2","slug":"benchmarking-large-language-models-on-2","title":"Benchmarking Large Language Models on Answering and Explaining Challenging Medical Questions","date":"2024-02-28","arxiv_id":"2402.18060","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-large-language-models-on-2#ran","syntology_url":"https://syntology.ai/paper/2402.18060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18060"}},"official":{"repos":["hanjiechen/challengeclinicalqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-data-science-agents","slug":"benchmarking-data-science-agents","title":"Benchmarking Data Science Agents","date":"2024-02-27","arxiv_id":"2402.17168","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-data-science-agents#ran","syntology_url":"https://syntology.ai/paper/2402.17168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17168"}},"official":{"repos":["metacopilot/dseval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-kandy-benchmark-incremental-neuro","slug":"the-kandy-benchmark-incremental-neuro","title":"The KANDY Benchmark: Incremental Neuro-Symbolic Learning and Reasoning with Kandinsky Patterns","date":"2024-02-27","arxiv_id":"2402.17431","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-kandy-benchmark-incremental-neuro#ran","syntology_url":"https://syntology.ai/paper/2402.17431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17431"}},"official":{"repos":["continual-nesy/kandybenchmark"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tombench-benchmarking-theory-of-mind-in-large","slug":"tombench-benchmarking-theory-of-mind-in-large","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","date":"2024-02-23","arxiv_id":"2402.15052","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tombench-benchmarking-theory-of-mind-in-large#ran","syntology_url":"https://syntology.ai/paper/2402.15052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15052"}},"official":{"repos":["zhchen18/tombench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/api-blend-a-comprehensive-corpora-for","slug":"api-blend-a-comprehensive-corpora-for","title":"API-BLEND: A Comprehensive Corpora for Training and Benchmarking API LLMs","date":"2024-02-23","arxiv_id":"2402.15491","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/api-blend-a-comprehensive-corpora-for#ran","syntology_url":"https://syntology.ai/paper/2402.15491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15491"}},"official":{"repos":["ibm/api-blend"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/criticbench-benchmarking-llms-for-critique","slug":"criticbench-benchmarking-llms-for-critique","title":"CriticBench: Benchmarking LLMs for Critique-Correct Reasoning","date":"2024-02-22","arxiv_id":"2402.14809","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/criticbench-benchmarking-llms-for-critique#ran","syntology_url":"https://syntology.ai/paper/2402.14809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14809"}},"official":{"repos":["CriticBench/CriticBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-llm-as-a-judge-robust-investigating","slug":"is-llm-as-a-judge-robust-investigating","title":"Is LLM-as-a-Judge Robust? Investigating Universal Adversarial Attacks on Zero-shot LLM Assessment","date":"2024-02-21","arxiv_id":"2402.14016","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/is-llm-as-a-judge-robust-investigating#ran","syntology_url":"https://syntology.ai/paper/2402.14016","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14016"}},"official":{"repos":["rainavyas/attack-comparative-assessment"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-soc-benchmarking-multimodal-large-language","slug":"mm-soc-benchmarking-multimodal-large-language","title":"MM-Soc: Benchmarking Multimodal Large Language Models in Social Media Platforms","date":"2024-02-21","arxiv_id":"2402.14154","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-soc-benchmarking-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.14154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14154"}},"official":{"repos":["claws-lab/mmsoc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-retrieval-augmented-generation","slug":"benchmarking-retrieval-augmented-generation","title":"Benchmarking Retrieval-Augmented Generation for Medicine","date":"2024-02-20","arxiv_id":"2402.13178","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2402.13178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13178"}},"official":{"repos":["teddy-xionggz/medrag","teddy-xionggz/mirage"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/class-incremental-learning-for-time-series","slug":"class-incremental-learning-for-time-series","title":"Class-incremental Learning for Time Series: Benchmark and Evaluation","date":"2024-02-19","arxiv_id":"2402.12035","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/class-incremental-learning-for-time-series#ran","syntology_url":"https://syntology.ai/paper/2402.12035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12035"}},"official":{"repos":["zqiao11/tscil"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analobench-benchmarking-the-identification-of","slug":"analobench-benchmarking-the-identification-of","title":"AnaloBench: Benchmarking the Identification of Abstract and Long-context Analogies","date":"2024-02-19","arxiv_id":"2402.12370","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/analobench-benchmarking-the-identification-of#ran","syntology_url":"https://syntology.ai/paper/2402.12370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12370"}},"official":{"repos":["jhu-clsp/analogical-reasoning","JHU-CLSP/AnaloBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/causalgym-benchmarking-causal","slug":"causalgym-benchmarking-causal","title":"CausalGym: Benchmarking causal interpretability methods on linguistic tasks","date":"2024-02-19","arxiv_id":"2402.12560","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causalgym-benchmarking-causal#ran","syntology_url":"https://syntology.ai/paper/2402.12560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12560"}},"official":{"repos":["aryamanarora/causalgym"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-knowledge-boundary-for-large","slug":"benchmarking-knowledge-boundary-for-large","title":"Benchmarking Knowledge Boundary for Large Language Models: A Different Perspective on Model Evaluation","date":"2024-02-18","arxiv_id":"2402.11493","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-knowledge-boundary-for-large#ran","syntology_url":"https://syntology.ai/paper/2402.11493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11493"}},"official":{"repos":["pkulcwmzx/knowledge-boundary"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-zeroth-order-optimization-for","slug":"revisiting-zeroth-order-optimization-for","title":"Revisiting Zeroth-Order Optimization for Memory-Efficient LLM Fine-Tuning: A Benchmark","date":"2024-02-18","arxiv_id":"2402.11592","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":3,"n_instrument":6,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/revisiting-zeroth-order-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2402.11592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11592"}},"official":{"repos":["zo-bench/zo-llm"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multimedeval-a-benchmark-and-a-toolkit-for","slug":"multimedeval-a-benchmark-and-a-toolkit-for","title":"MultiMedEval: A Benchmark and a Toolkit for Evaluating Medical Vision-Language Models","date":"2024-02-14","arxiv_id":"2402.09262","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimedeval-a-benchmark-and-a-toolkit-for#ran","syntology_url":"https://syntology.ai/paper/2402.09262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09262"}},"official":{"repos":["corentin-ryr/multimedeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/massively-multi-cultural-knowledge","slug":"massively-multi-cultural-knowledge","title":"Massively Multi-Cultural Knowledge Acquisition & LM Benchmarking","date":"2024-02-14","arxiv_id":"2402.09369","repositories_listed":1,"syntology":{"n":14,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/massively-multi-cultural-knowledge#ran","syntology_url":"https://syntology.ai/paper/2402.09369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09369"}},"official":{"repos":["yrf1/llm-massivemulticulturenormsknowledge-nclb"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/lota-bench-benchmarking-language-oriented","slug":"lota-bench-benchmarking-language-oriented","title":"LoTa-Bench: Benchmarking Language-oriented Task Planners for Embodied Agents","date":"2024-02-13","arxiv_id":"2402.08178","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lota-bench-benchmarking-language-oriented#ran","syntology_url":"https://syntology.ai/paper/2402.08178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08178"}},"official":{"repos":["lbaa2022/llmtaskplanning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/customizable-perturbation-synthesis-for","slug":"customizable-perturbation-synthesis-for","title":"Customizable Perturbation Synthesis for Robust SLAM Benchmarking","date":"2024-02-12","arxiv_id":"2402.08125","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/customizable-perturbation-synthesis-for#ran","syntology_url":"https://syntology.ai/paper/2402.08125","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08125"}},"official":{"repos":["xiaohao-xu/slam-under-perturbation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-diffusion-models-for-amortized-inference","slug":"on-diffusion-models-for-amortized-inference","title":"Improved off-policy training of diffusion samplers","date":"2024-02-07","arxiv_id":"2402.05098","repositories_listed":1,"syntology":{"n":17,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":17,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-diffusion-models-for-amortized-inference#ran","syntology_url":"https://syntology.ai/paper/2402.05098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05098"}},"official":{"repos":["gfnorg/gfn-diffusion"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lv-eval-a-balanced-long-context-benchmark","slug":"lv-eval-a-balanced-long-context-benchmark","title":"LV-Eval: A Balanced Long-Context Benchmark with 5 Length Levels Up to 256K","date":"2024-02-06","arxiv_id":"2402.05136","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lv-eval-a-balanced-long-context-benchmark#ran","syntology_url":"https://syntology.ai/paper/2402.05136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05136"}},"official":{"repos":["infinigence/lveval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/ltu-ili-an-all-in-one-framework-for-implicit","slug":"ltu-ili-an-all-in-one-framework-for-implicit","title":"LtU-ILI: An All-in-One Framework for Implicit Inference in Astrophysics and Cosmology","date":"2024-02-06","arxiv_id":"2402.05137","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ltu-ili-an-all-in-one-framework-for-implicit#ran","syntology_url":"https://syntology.ai/paper/2402.05137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05137"}},"official":{"repos":["maho3/ltu-ili"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genface-a-large-scale-fine-grained-face","slug":"genface-a-large-scale-fine-grained-face","title":"GenFace: A Large-Scale Fine-Grained Face Forgery Benchmark and Cross Appearance-Edge Learning","date":"2024-02-03","arxiv_id":"2402.02003","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":1,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 2 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genface-a-large-scale-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2402.02003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02003"}},"official":{"repos":["jenine-321/genface"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/effibench-benchmarking-the-efficiency-of","slug":"effibench-benchmarking-the-efficiency-of","title":"EffiBench: Benchmarking the Efficiency of Automatically Generated Code","date":"2024-02-03","arxiv_id":"2402.02037","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/effibench-benchmarking-the-efficiency-of#ran","syntology_url":"https://syntology.ai/paper/2402.02037","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02037"}},"official":{"repos":["huangd1999/EffiBench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/we-re-not-using-videos-effectively-an-updated","slug":"we-re-not-using-videos-effectively-an-updated","title":"We're Not Using Videos Effectively: An Updated Domain Adaptive Video Segmentation Baseline","date":"2024-02-01","arxiv_id":"2402.00868","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/we-re-not-using-videos-effectively-an-updated#ran","syntology_url":"https://syntology.ai/paper/2402.00868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00868"}},"official":{"repos":["simarkareer/unifiedvideoda"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/good-at-captioning-bad-at-counting","slug":"good-at-captioning-bad-at-counting","title":"Good at captioning, bad at counting: Benchmarking GPT-4V on Earth observation data","date":"2024-01-31","arxiv_id":"2401.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/good-at-captioning-bad-at-counting#ran","syntology_url":"https://syntology.ai/paper/2401.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17600"}},"official":{"repos":["Earth-Intelligence-Lab/vleo-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/i-think-therefore-i-am-awareness-in-large","slug":"i-think-therefore-i-am-awareness-in-large","title":"I Think, Therefore I am: Benchmarking Awareness of Large Language Models Using AwareBench","date":"2024-01-31","arxiv_id":"2401.17882","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/i-think-therefore-i-am-awareness-in-large#ran","syntology_url":"https://syntology.ai/paper/2401.17882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17882"}},"official":{"repos":["howiehwong/awareness-in-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/ppm-automated-generation-of-diverse","slug":"ppm-automated-generation-of-diverse","title":"PPM: Automated Generation of Diverse Programming Problems for Benchmarking Code Generation Models","date":"2024-01-28","arxiv_id":"2401.15545","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ppm-automated-generation-of-diverse#ran","syntology_url":"https://syntology.ai/paper/2401.15545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15545"}},"official":{"repos":["seekingdream/ppm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multihop-rag-benchmarking-retrieval-augmented","slug":"multihop-rag-benchmarking-retrieval-augmented","title":"MultiHop-RAG: Benchmarking Retrieval-Augmented Generation for Multi-Hop Queries","date":"2024-01-27","arxiv_id":"2401.15391","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multihop-rag-benchmarking-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2401.15391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15391"}},"official":{"repos":["yixuantt/MultiHop-RAG"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/agentboard-an-analytical-evaluation-board-of","slug":"agentboard-an-analytical-evaluation-board-of","title":"AgentBoard: An Analytical Evaluation Board of Multi-turn LLM Agents","date":"2024-01-24","arxiv_id":"2401.13178","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentboard-an-analytical-evaluation-board-of#ran","syntology_url":"https://syntology.ai/paper/2401.13178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13178"}},"official":{"repos":["hkust-nlp/agentboard"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scimmir-benchmarking-scientific-multi-modal","slug":"scimmir-benchmarking-scientific-multi-modal","title":"SciMMIR: Benchmarking Scientific Multi-modal Information Retrieval","date":"2024-01-24","arxiv_id":"2401.13478","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scimmir-benchmarking-scientific-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2401.13478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13478"}},"official":{"repos":["wusiwei0410/scimmir"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-neural-network-benchmarks-for-selective","slug":"deep-neural-network-benchmarks-for-selective","title":"Deep Neural Network Benchmarks for Selective Classification","date":"2024-01-23","arxiv_id":"2401.12708","repositories_listed":1,"syntology":{"n":15,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":15,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/deep-neural-network-benchmarks-for-selective#ran","syntology_url":"https://syntology.ai/paper/2401.12708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12708"}},"official":{"repos":["andrepugni/esc"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llms-via-uncertainty","slug":"benchmarking-llms-via-uncertainty","title":"Benchmarking LLMs via Uncertainty Quantification","date":"2024-01-23","arxiv_id":"2401.12794","repositories_listed":1,"syntology":{"n":17,"n_ran":13,"n_constructed":0,"n_ran_checked":1,"n_instrument":12,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 12 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-llms-via-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2401.12794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12794"}},"official":{"repos":["smartyfh/llm-uncertainty-bench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-multimodal-models-against","slug":"benchmarking-large-multimodal-models-against","title":"Benchmarking Large Multimodal Models against Common Corruptions","date":"2024-01-22","arxiv_id":"2401.11943","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-multimodal-models-against#ran","syntology_url":"https://syntology.ai/paper/2401.11943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11943"}},"official":{"repos":["sail-sg/mmcbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/r-judge-benchmarking-safety-risk-awareness","slug":"r-judge-benchmarking-safety-risk-awareness","title":"R-Judge: Benchmarking Safety Risk Awareness for LLM Agents","date":"2024-01-18","arxiv_id":"2401.10019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r-judge-benchmarking-safety-risk-awareness#ran","syntology_url":"https://syntology.ai/paper/2401.10019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10019"}},"official":{"repos":["lordog/r-judge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-the-robustness-of-image","slug":"benchmarking-the-robustness-of-image","title":"WAVES: Benchmarking the Robustness of Image Watermarks","date":"2024-01-16","arxiv_id":"2401.08573","repositories_listed":1,"syntology":{"n":21,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":21,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/benchmarking-the-robustness-of-image#ran","syntology_url":"https://syntology.ai/paper/2401.08573","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08573"}},"official":{"repos":["umd-huang-lab/waves"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/rsud20k-a-dataset-for-road-scene","slug":"rsud20k-a-dataset-for-road-scene","title":"RSUD20K: A Dataset for Road Scene Understanding In Autonomous Driving","date":"2024-01-14","arxiv_id":"2401.07322","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rsud20k-a-dataset-for-road-scene#ran","syntology_url":"https://syntology.ai/paper/2401.07322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07322"}},"official":{"repos":["hasibzunair/rsud20k"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/infiagent-dabench-evaluating-agents-on-data","slug":"infiagent-dabench-evaluating-agents-on-data","title":"InfiAgent-DABench: Evaluating Agents on Data Analysis Tasks","date":"2024-01-10","arxiv_id":"2401.05507","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/infiagent-dabench-evaluating-agents-on-data#ran","syntology_url":"https://syntology.ai/paper/2401.05507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05507"}},"official":{"repos":["infiagent/infiagent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on","slug":"benchmarking-large-language-models-on","title":"Benchmarking Large Language Models on Controllable Generation under Diversified Instructions","date":"2024-01-01","arxiv_id":"2401.00690","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2401.00690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00690"}},"official":{"repos":["xt-cyh/codi-eval"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-defending-against-indirect","slug":"benchmarking-and-defending-against-indirect","title":"Benchmarking and Defending Against Indirect Prompt Injection Attacks on Large Language Models","date":"2023-12-21","arxiv_id":"2312.14197","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-and-defending-against-indirect#ran","syntology_url":"https://syntology.ai/paper/2312.14197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14197"}},"official":{"repos":["microsoft/BIPIA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-compute-is-not-all-you-need-for","slug":"scaling-compute-is-not-all-you-need-for","title":"Scaling Compute Is Not All You Need for Adversarial Robustness","date":"2023-12-20","arxiv_id":"2312.13131","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":7,"n_pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 3 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/scaling-compute-is-not-all-you-need-for#ran","syntology_url":"https://syntology.ai/paper/2312.13131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13131"}},"official":{"repos":["dedeswim/timm-adv-training"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fifar-a-fraud-detection-dataset-for-learning","slug":"fifar-a-fraud-detection-dataset-for-learning","title":"FiFAR: A Fraud Detection Dataset for Learning to Defer","date":"2023-12-20","arxiv_id":"2312.13218","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fifar-a-fraud-detection-dataset-for-learning#ran","syntology_url":"https://syntology.ai/paper/2312.13218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13218"}},"official":{"repos":["feedzai/fifar-dataset"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-train-neural-field-representations-a","slug":"how-to-train-neural-field-representations-a","title":"How to Train Neural Field Representations: A Comprehensive Study and Benchmark","date":"2023-12-16","arxiv_id":"2312.10531","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/how-to-train-neural-field-representations-a#ran","syntology_url":"https://syntology.ai/paper/2312.10531","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10531"}},"official":{"repos":["samuelepapa/fit-a-nef"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-well-does-gpt-4v-ision-adapt-to","slug":"how-well-does-gpt-4v-ision-adapt-to","title":"How Well Does GPT-4V(ision) Adapt to Distribution Shifts? A Preliminary Investigation","date":"2023-12-12","arxiv_id":"2312.07424","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-well-does-gpt-4v-ision-adapt-to#ran","syntology_url":"https://syntology.ai/paper/2312.07424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07424"}},"official":{"repos":["jameszhou-gl/gpt-4v-distribution-shift"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/eq-bench-an-emotional-intelligence-benchmark","slug":"eq-bench-an-emotional-intelligence-benchmark","title":"EQ-Bench: An Emotional Intelligence Benchmark for Large Language Models","date":"2023-12-11","arxiv_id":"2312.06281","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eq-bench-an-emotional-intelligence-benchmark#ran","syntology_url":"https://syntology.ai/paper/2312.06281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06281"}},"official":{"repos":["eq-bench/eq-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egoplan-bench-benchmarking-egocentric","slug":"egoplan-bench-benchmarking-egocentric","title":"EgoPlan-Bench: Benchmarking Multimodal Large Language Models for Human-Level Planning","date":"2023-12-11","arxiv_id":"2312.06722","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/egoplan-bench-benchmarking-egocentric#ran","syntology_url":"https://syntology.ai/paper/2312.06722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06722"}},"official":{"repos":["chenyi99/egoplan"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/am-radio-agglomerative-model-reduce-all","slug":"am-radio-agglomerative-model-reduce-all","title":"AM-RADIO: Agglomerative Vision Foundation Model -- Reduce All Domains Into One","date":"2023-12-10","arxiv_id":"2312.06709","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/am-radio-agglomerative-model-reduce-all#ran","syntology_url":"https://syntology.ai/paper/2312.06709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06709"}},"official":{"repos":["nvlabs/radio"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/benchmarking-distribution-shift-in-tabular-1","slug":"benchmarking-distribution-shift-in-tabular-1","title":"Benchmarking Distribution Shift in Tabular Data with TableShift","date":"2023-12-10","arxiv_id":"2312.07577","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-distribution-shift-in-tabular-1#ran","syntology_url":"https://syntology.ai/paper/2312.07577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07577"}},"official":{"repos":["mlfoundations/tableshift"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-analysis-of-unsupervised","slug":"benchmarking-and-analysis-of-unsupervised","title":"Benchmarking and Analysis of Unsupervised Object Segmentation from Real-world Single Images","date":"2023-12-08","arxiv_id":"2312.04947","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-analysis-of-unsupervised#ran","syntology_url":"https://syntology.ai/paper/2312.04947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04947"}},"official":{"repos":["vlar-group/unsupobjseg"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bedd-the-minerl-basalt-evaluation-and-1","slug":"bedd-the-minerl-basalt-evaluation-and-1","title":"BEDD: The MineRL BASALT Evaluation and Demonstrations Dataset for Training and Benchmarking Agents that Solve Fuzzy Tasks","date":"2023-12-05","arxiv_id":"2312.02405","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bedd-the-minerl-basalt-evaluation-and-1#ran","syntology_url":"https://syntology.ai/paper/2312.02405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02405"}},"official":{"repos":["minerllabs/basalt-benchmark"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchlmm-benchmarking-cross-style-visual","slug":"benchlmm-benchmarking-cross-style-visual","title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","date":"2023-12-05","arxiv_id":"2312.02896","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchlmm-benchmarking-cross-style-visual#ran","syntology_url":"https://syntology.ai/paper/2312.02896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02896"}},"official":{"repos":["aifeg/benchgpt","aifeg/benchlmm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarl-benchmarking-multi-agent","slug":"benchmarl-benchmarking-multi-agent","title":"BenchMARL: Benchmarking Multi-Agent Reinforcement Learning","date":"2023-12-03","arxiv_id":"2312.01472","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarl-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2312.01472","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01472"}},"official":{"repos":["facebookresearch/benchmarl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alignbench-benchmarking-chinese-alignment-of","slug":"alignbench-benchmarking-chinese-alignment-of","title":"AlignBench: Benchmarking Chinese Alignment of Large Language Models","date":"2023-11-30","arxiv_id":"2311.18743","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/alignbench-benchmarking-chinese-alignment-of#ran","syntology_url":"https://syntology.ai/paper/2311.18743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18743"}},"official":{"repos":["thudm/alignbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-ligand-pose-sampling-for-molecular","slug":"enhancing-ligand-pose-sampling-for-molecular","title":"Enhancing Ligand Pose Sampling for Molecular Docking","date":"2023-11-30","arxiv_id":"2312.00191","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-ligand-pose-sampling-for-molecular#ran","syntology_url":"https://syntology.ai/paper/2312.00191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.00191"}},"official":{"repos":["drorlab/glow_ives"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/are-we-going-mad-benchmarking-multi-agent","slug":"are-we-going-mad-benchmarking-multi-agent","title":"Should we be going MAD? A Look at Multi-Agent Debate Strategies for LLMs","date":"2023-11-29","arxiv_id":"2311.17371","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-we-going-mad-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2311.17371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17371"}},"official":{"repos":["instadeepai/debatellm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uhgeval-benchmarking-the-hallucination-of","slug":"uhgeval-benchmarking-the-hallucination-of","title":"UHGEval: Benchmarking the Hallucination of Chinese Large Language Models via Unconstrained Generation","date":"2023-11-26","arxiv_id":"2311.15296","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uhgeval-benchmarking-the-hallucination-of#ran","syntology_url":"https://syntology.ai/paper/2311.15296","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15296"}},"official":{"repos":["IAAR-Shanghai/UHGEval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-robustness-of-text-image","slug":"benchmarking-robustness-of-text-image","title":"Benchmarking Robustness of Text-Image Composed Retrieval","date":"2023-11-24","arxiv_id":"2311.14837","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-robustness-of-text-image#ran","syntology_url":"https://syntology.ai/paper/2311.14837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.14837"}},"official":{"repos":["suntongtongtong/benchmark-robustness-text-image-compose-retrieval"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pg-video-llava-pixel-grounding-large-video","slug":"pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","arxiv_id":"2311.13435","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pg-video-llava-pixel-grounding-large-video#ran","syntology_url":"https://syntology.ai/paper/2311.13435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13435"}},"official":{"repos":["mbzuai-oryx/video-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bend-benchmarking-dna-language-models-on","slug":"bend-benchmarking-dna-language-models-on","title":"BEND: Benchmarking DNA Language Models on biologically meaningful tasks","date":"2023-11-21","arxiv_id":"2311.12570","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/bend-benchmarking-dna-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2311.12570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.12570"}},"official":{"repos":["frederikkemarin/bend"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}}],"record_sha256":"78eb09e304b3658bb2939492050dbc7287373bc950901d7f99588b1e81b141ad","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}