{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/ran/2","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":8,"rows_per_page":100,"rows":[101,200],"of":749,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking/papers/ran/1","prev":"/task/benchmarking/papers/ran/1","next":"/task/benchmarking/papers/ran/3","papers":[{"url":"/paper/stackeval-benchmarking-llms-in-coding","slug":"stackeval-benchmarking-llms-in-coding","title":"StackEval: Benchmarking LLMs in Coding Assistance","date":"2024-11-21","arxiv_id":"2412.05288","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stackeval-benchmarking-llms-in-coding#ran","syntology_url":"https://syntology.ai/paper/2412.05288","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05288"}},"official":{"repos":["ProsusAI/stack-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/delta-influence-unlearning-poisons-via","slug":"delta-influence-unlearning-poisons-via","title":"Delta-Influence: Unlearning Poisons via Influence Functions","date":"2024-11-20","arxiv_id":"2411.13731","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/delta-influence-unlearning-poisons-via#ran","syntology_url":"https://syntology.ai/paper/2411.13731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13731"}},"official":{"repos":["andyisokay/delta-influence"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/benchmarking-positional-encodings-for-gnns","slug":"benchmarking-positional-encodings-for-gnns","title":"Benchmarking Positional Encodings for GNNs and Graph Transformers","date":"2024-11-19","arxiv_id":"2411.12732","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-positional-encodings-for-gnns#ran","syntology_url":"https://syntology.ai/paper/2411.12732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.12732"}},"official":{"repos":["ETH-DISCO/Benchmarking-PEs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fm-ts-flow-matching-for-time-series","slug":"fm-ts-flow-matching-for-time-series","title":"FM-TS: Flow Matching for Time Series Generation","date":"2024-11-12","arxiv_id":"2411.07506","repositories_listed":1,"syntology":{"n":20,"n_ran":20,"n_constructed":0,"n_ran_checked":16,"n_instrument":4,"n_unverified":0,"n_honours":2,"n_violates":3,"n_no_contract":11,"n_pointer_only":20,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 3 violated, 11 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fm-ts-flow-matching-for-time-series#ran","syntology_url":"https://syntology.ai/paper/2411.07506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07506"}},"official":{"repos":["unites-lab/fmts"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieval-or-global-context-understanding-on","slug":"retrieval-or-global-context-understanding-on","title":"Retrieval or Global Context Understanding? On Many-Shot In-Context Learning for Long-Context Evaluation","date":"2024-11-11","arxiv_id":"2411.07130","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retrieval-or-global-context-understanding-on#ran","syntology_url":"https://syntology.ai/paper/2411.07130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.07130"}},"official":{"repos":["launchnlp/ManyICLBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-distributional-alignment-of","slug":"benchmarking-distributional-alignment-of","title":"Benchmarking Distributional Alignment of Large Language Models","date":"2024-11-08","arxiv_id":"2411.05403","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-distributional-alignment-of#ran","syntology_url":"https://syntology.ai/paper/2411.05403","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05403"}},"official":{"repos":["nicolemeister/benchmarking-distributional-alignment"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hourvideo-1-hour-video-language-understanding","slug":"hourvideo-1-hour-video-language-understanding","title":"HourVideo: 1-Hour Video-Language Understanding","date":"2024-11-07","arxiv_id":"2411.04998","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hourvideo-1-hour-video-language-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.04998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04998"}},"official":{"repos":["keshik6/HourVideo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-loss-of-context-awareness-in-general","slug":"on-the-loss-of-context-awareness-in-general","title":"On the Loss of Context-awareness in General Instruction Fine-tuning","date":"2024-11-05","arxiv_id":"2411.02688","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-loss-of-context-awareness-in-general#ran","syntology_url":"https://syntology.ai/paper/2411.02688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02688"}},"official":{"repos":["YihanWang617/context_awareness"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interaction2code-how-far-are-we-from","slug":"interaction2code-how-far-are-we-from","title":"Interaction2Code: Benchmarking MLLM-based Interactive Webpage Code Generation from Interactive Prototyping","date":"2024-11-05","arxiv_id":"2411.03292","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/interaction2code-how-far-are-we-from#ran","syntology_url":"https://syntology.ai/paper/2411.03292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03292"}},"official":{"repos":["webpai/interaction2code"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tddbench-a-benchmark-for-training-data","slug":"tddbench-a-benchmark-for-training-data","title":"TDDBench: A Benchmark for Training data detection","date":"2024-11-05","arxiv_id":"2411.03363","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tddbench-a-benchmark-for-training-data#ran","syntology_url":"https://syntology.ai/paper/2411.03363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03363"}},"official":null}},{"url":"/paper/benchmarking-vision-language-model-unlearning","slug":"benchmarking-vision-language-model-unlearning","title":"Benchmarking Vision Language Model Unlearning via Fictitious Facial Identity Dataset","date":"2024-11-05","arxiv_id":"2411.03554","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-vision-language-model-unlearning#ran","syntology_url":"https://syntology.ai/paper/2411.03554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03554"}},"official":{"repos":["safolab-wisc/fiubench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/tablegpt2-a-large-multimodal-model-with","slug":"tablegpt2-a-large-multimodal-model-with","title":"TableGPT2: A Large Multimodal Model with Tabular Data Integration","date":"2024-11-04","arxiv_id":"2411.02059","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tablegpt2-a-large-multimodal-model-with#ran","syntology_url":"https://syntology.ai/paper/2411.02059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02059"}},"official":{"repos":["tablegpt/tablegpt-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/layerdag-a-layerwise-autoregressive-diffusion","slug":"layerdag-a-layerwise-autoregressive-diffusion","title":"LayerDAG: A Layerwise Autoregressive Diffusion Model for Directed Acyclic Graph Generation","date":"2024-11-04","arxiv_id":"2411.02322","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/layerdag-a-layerwise-autoregressive-diffusion#ran","syntology_url":"https://syntology.ai/paper/2411.02322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02322"}},"official":{"repos":["graph-com/layerdag"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/road-waymo-action-awareness-at-scale-for","slug":"road-waymo-action-awareness-at-scale-for","title":"ROAD-Waymo: Action Awareness at Scale for Autonomous Driving","date":"2024-11-03","arxiv_id":"2411.01683","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/road-waymo-action-awareness-at-scale-for#ran","syntology_url":"https://syntology.ai/paper/2411.01683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.01683"}},"official":{"repos":["salmank255/ROAD_Waymo_Baseline"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/detectrl-benchmarking-llm-generated-text","slug":"detectrl-benchmarking-llm-generated-text","title":"DetectRL: Benchmarking LLM-Generated Text Detection in Real-World Scenarios","date":"2024-10-31","arxiv_id":"2410.23746","repositories_listed":1,"syntology":{"n":16,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":11,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":16,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/detectrl-benchmarking-llm-generated-text#ran","syntology_url":"https://syntology.ai/paper/2410.23746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23746"}},"official":{"repos":["nlp2ct/detectrl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/cale-continuous-arcade-learning-environment","slug":"cale-continuous-arcade-learning-environment","title":"CALE: Continuous Arcade Learning Environment","date":"2024-10-31","arxiv_id":"2410.23810","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cale-continuous-arcade-learning-environment#ran","syntology_url":"https://syntology.ai/paper/2410.23810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23810"}},"official":{"repos":["farama-foundation/arcade-learning-environment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/allclear-a-comprehensive-dataset-and","slug":"allclear-a-comprehensive-dataset-and","title":"AllClear: A Comprehensive Dataset and Benchmark for Cloud Removal in Satellite Imagery","date":"2024-10-31","arxiv_id":"2410.23891","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/allclear-a-comprehensive-dataset-and#ran","syntology_url":"https://syntology.ai/paper/2410.23891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23891"}},"official":{"repos":["zhou-hangyu/allclear"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/androidlab-training-and-systematic","slug":"androidlab-training-and-systematic","title":"AndroidLab: Training and Systematic Benchmarking of Android Autonomous Agents","date":"2024-10-31","arxiv_id":"2410.24024","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/androidlab-training-and-systematic#ran","syntology_url":"https://syntology.ai/paper/2410.24024","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.24024"}},"official":{"repos":["THUDM/Android-Lab"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-inference-bench-inference-benchmarking-of","slug":"llm-inference-bench-inference-benchmarking-of","title":"LLM-Inference-Bench: Inference Benchmarking of Large Language Models on AI Accelerators","date":"2024-10-31","arxiv_id":"2411.00136","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-inference-bench-inference-benchmarking-of#ran","syntology_url":"https://syntology.ai/paper/2411.00136","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00136"}},"official":{"repos":["argonne-lcf/llm-inference-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm4mat-bench-benchmarking-large-language","slug":"llm4mat-bench-benchmarking-large-language","title":"LLM4Mat-Bench: Benchmarking Large Language Models for Materials Property Prediction","date":"2024-10-31","arxiv_id":"2411.00177","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm4mat-bench-benchmarking-large-language#ran","syntology_url":"https://syntology.ai/paper/2411.00177","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00177"}},"official":{"repos":["vertaix/llm4mat-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/injecguard-benchmarking-and-mitigating-over","slug":"injecguard-benchmarking-and-mitigating-over","title":"InjecGuard: Benchmarking and Mitigating Over-defense in Prompt Injection Guardrail Models","date":"2024-10-30","arxiv_id":"2410.22770","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/injecguard-benchmarking-and-mitigating-over#ran","syntology_url":"https://syntology.ai/paper/2410.22770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22770"}},"official":{"repos":["leolee99/injecguard","safolab-wisc/injecguard"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/odrl-a-benchmark-for-off-dynamics","slug":"odrl-a-benchmark-for-off-dynamics","title":"ODRL: A Benchmark for Off-Dynamics Reinforcement Learning","date":"2024-10-28","arxiv_id":"2410.20750","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/odrl-a-benchmark-for-off-dynamics#ran","syntology_url":"https://syntology.ai/paper/2410.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20750"}},"official":{"repos":["offdynamicsrl/off-dynamics-rl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/codes-benchmarking-coupled-ode-surrogates","slug":"codes-benchmarking-coupled-ode-surrogates","title":"CODES: Benchmarking Coupled ODE Surrogates","date":"2024-10-28","arxiv_id":"2410.20886","repositories_listed":1,"syntology":{"n":19,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":19,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/codes-benchmarking-coupled-ode-surrogates#ran","syntology_url":"https://syntology.ai/paper/2410.20886","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20886"}},"official":{"repos":["robin-janssen/codes-benchmark"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/autobench-v-can-large-vision-language-models","slug":"autobench-v-can-large-vision-language-models","title":"AutoBench-V: Can Large Vision-Language Models Benchmark Themselves?","date":"2024-10-28","arxiv_id":"2410.21259","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autobench-v-can-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2410.21259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21259"}},"official":{"repos":["wad3birch/AutoBench-V"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llmcbench-benchmarking-large-language-model","slug":"llmcbench-benchmarking-large-language-model","title":"LLMCBench: Benchmarking Large Language Model Compression for Efficient Deployment","date":"2024-10-28","arxiv_id":"2410.21352","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llmcbench-benchmarking-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2410.21352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21352"}},"official":{"repos":["aboveparadise/llmcbench"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-auditing-test-to-detect-behavioral-shift","slug":"an-auditing-test-to-detect-behavioral-shift","title":"An Auditing Test To Detect Behavioral Shift in Language Models","date":"2024-10-25","arxiv_id":"2410.19406","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/an-auditing-test-to-detect-behavioral-shift#ran","syntology_url":"https://syntology.ai/paper/2410.19406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19406"}},"official":{"repos":["richterleo/Auditing_Test_for_LMs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-watermarking-using-generative-priors","slug":"robust-watermarking-using-generative-priors","title":"Robust Watermarking Using Generative Priors Against Image Editing: From Benchmarking to Advances","date":"2024-10-24","arxiv_id":"2410.18775","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-watermarking-using-generative-priors#ran","syntology_url":"https://syntology.ai/paper/2410.18775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18775"}},"official":{"repos":["shilin-lu/vine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/conditional-diffusions-for-neural-posterior","slug":"conditional-diffusions-for-neural-posterior","title":"Conditional diffusions for amortized neural posterior estimation","date":"2024-10-24","arxiv_id":"2410.19105","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conditional-diffusions-for-neural-posterior#ran","syntology_url":"https://syntology.ai/paper/2410.19105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19105"}},"official":{"repos":["tianyucodings/cdiff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/voicebench-benchmarking-llm-based-voice","slug":"voicebench-benchmarking-llm-based-voice","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","date":"2024-10-22","arxiv_id":"2410.17196","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/voicebench-benchmarking-llm-based-voice#ran","syntology_url":"https://syntology.ai/paper/2410.17196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17196"}},"official":{"repos":["matthewcym/voicebench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-if-benchmarking-llms-on-multi-turn-and","slug":"multi-if-benchmarking-llms-on-multi-turn-and","title":"Multi-IF: Benchmarking LLMs on Multi-Turn and Multilingual Instructions Following","date":"2024-10-21","arxiv_id":"2410.15553","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-if-benchmarking-llms-on-multi-turn-and#ran","syntology_url":"https://syntology.ai/paper/2410.15553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15553"}},"official":{"repos":["facebookresearch/Multi-IF"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-pathology-foundation-models","slug":"benchmarking-pathology-foundation-models","title":"Benchmarking Pathology Foundation Models: Adaptation Strategies and Scenarios","date":"2024-10-21","arxiv_id":"2410.16038","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":15,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-pathology-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2410.16038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16038"}},"official":{"repos":["quiil/benchmarkingpathologyfoundationmodels"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/spa-bench-a-comprehensive-benchmark-for","slug":"spa-bench-a-comprehensive-benchmark-for","title":"SPA-Bench: A Comprehensive Benchmark for SmartPhone Agent Evaluation","date":"2024-10-19","arxiv_id":"2410.15164","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/spa-bench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.15164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15164"}},"official":{"repos":["ai-agents-2030/SPA-Bench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/intersectionzoo-eco-driving-for-benchmarking","slug":"intersectionzoo-eco-driving-for-benchmarking","title":"IntersectionZoo: Eco-driving for Benchmarking Multi-Agent Contextual Reinforcement Learning","date":"2024-10-19","arxiv_id":"2410.15221","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intersectionzoo-eco-driving-for-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.15221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15221"}},"official":{"repos":["mit-wu-lab/IntersectionZoo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multichartqa-benchmarking-vision-language","slug":"multichartqa-benchmarking-vision-language","title":"MultiChartQA: Benchmarking Vision-Language Models on Multi-Chart Problems","date":"2024-10-18","arxiv_id":"2410.14179","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multichartqa-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.14179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14179"}},"official":{"repos":["zivenzhu/multi-chart-qa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-lingual-auto-evaluation-for-assessing","slug":"cross-lingual-auto-evaluation-for-assessing","title":"Cross-Lingual Auto Evaluation for Assessing Multilingual LLMs","date":"2024-10-17","arxiv_id":"2410.13394","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cross-lingual-auto-evaluation-for-assessing#ran","syntology_url":"https://syntology.ai/paper/2410.13394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13394"}},"official":{"repos":["ai4bharat/cia"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ab-initio-nonparametric-variable-selection","slug":"ab-initio-nonparametric-variable-selection","title":"Ab Initio Nonparametric Variable Selection for Scalable Symbolic Regression with Large $p$","date":"2024-10-17","arxiv_id":"2410.13681","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ab-initio-nonparametric-variable-selection#ran","syntology_url":"https://syntology.ai/paper/2410.13681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13681"}},"official":{"repos":["mattsheng/PAN_SR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-transcriptomics-foundation","slug":"benchmarking-transcriptomics-foundation","title":"Benchmarking Transcriptomics Foundation Models for Perturbation Analysis : one PCA still rules them all","date":"2024-10-17","arxiv_id":"2410.13956","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-transcriptomics-foundation#ran","syntology_url":"https://syntology.ai/paper/2410.13956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13956"}},"official":{"repos":["valence-labs/Tx-Evaluation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rclicks-realistic-click-simulation-for","slug":"rclicks-realistic-click-simulation-for","title":"RClicks: Realistic Click Simulation for Benchmarking Interactive Segmentation","date":"2024-10-15","arxiv_id":"2410.11722","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rclicks-realistic-click-simulation-for#ran","syntology_url":"https://syntology.ai/paper/2410.11722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11722"}},"official":{"repos":["emb-ai/rclicks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mlperf-power-benchmarking-the-energy","slug":"mlperf-power-benchmarking-the-energy","title":"MLPerf Power: Benchmarking the Energy Efficiency of Machine Learning Systems from Microwatts to Megawatts for Sustainable AI","date":"2024-10-15","arxiv_id":"2410.12032","repositories_listed":2,"syntology":{"n":13,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/mlperf-power-benchmarking-the-energy#ran","syntology_url":"https://syntology.ai/paper/2410.12032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.12032"}},"official":null}},{"url":"/paper/longmemeval-benchmarking-chat-assistants-on","slug":"longmemeval-benchmarking-chat-assistants-on","title":"LongMemEval: Benchmarking Chat Assistants on Long-Term Interactive Memory","date":"2024-10-14","arxiv_id":"2410.10813","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/longmemeval-benchmarking-chat-assistants-on#ran","syntology_url":"https://syntology.ai/paper/2410.10813","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10813"}},"official":{"repos":["xiaowu0162/longmemeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/temporalbench-benchmarking-fine-grained","slug":"temporalbench-benchmarking-fine-grained","title":"TemporalBench: Benchmarking Fine-grained Temporal Understanding for Multimodal Video Models","date":"2024-10-14","arxiv_id":"2410.10818","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/temporalbench-benchmarking-fine-grained#ran","syntology_url":"https://syntology.ai/paper/2410.10818","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10818"}},"official":{"repos":["mu-cai/TemporalBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rmb-comprehensively-benchmarking-reward","slug":"rmb-comprehensively-benchmarking-reward","title":"RMB: Comprehensively Benchmarking Reward Models in LLM Alignment","date":"2024-10-13","arxiv_id":"2410.09893","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rmb-comprehensively-benchmarking-reward#ran","syntology_url":"https://syntology.ai/paper/2410.09893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09893"}},"official":{"repos":["zhou-zoey/rmb-reward-model-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/fb-bench-a-fine-grained-multi-task-benchmark","slug":"fb-bench-a-fine-grained-multi-task-benchmark","title":"FB-Bench: A Fine-Grained Multi-Task Benchmark for Evaluating LLMs' Responsiveness to Human Feedback","date":"2024-10-12","arxiv_id":"2410.09412","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fb-bench-a-fine-grained-multi-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2410.09412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09412"}},"official":{"repos":["pku-baichuan-mlsystemlab/fb-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-agentic-workflow-generation","slug":"benchmarking-agentic-workflow-generation","title":"Benchmarking Agentic Workflow Generation","date":"2024-10-10","arxiv_id":"2410.07869","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-agentic-workflow-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07869"}},"official":{"repos":["zjunlp/worfbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compl-ai-framework-a-technical-interpretation","slug":"compl-ai-framework-a-technical-interpretation","title":"COMPL-AI Framework: A Technical Interpretation and LLM Benchmarking Suite for the EU Artificial Intelligence Act","date":"2024-10-10","arxiv_id":"2410.07959","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compl-ai-framework-a-technical-interpretation#ran","syntology_url":"https://syntology.ai/paper/2410.07959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07959"}},"official":{"repos":["compl-ai/compl-ai"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embodied-agent-interface-benchmarking-llms","slug":"embodied-agent-interface-benchmarking-llms","title":"Embodied Agent Interface: Benchmarking LLMs for Embodied Decision Making","date":"2024-10-09","arxiv_id":"2410.07166","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embodied-agent-interface-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2410.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07166"}},"official":{"repos":["embodied-agent-eval/embodied-agent-eval","embodied-agent-interface/embodied-agent-interface","embodied-agent-eval/embodied-agent-eval.github.io"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-data-heterogeneity-evaluation","slug":"benchmarking-data-heterogeneity-evaluation","title":"Benchmarking Data Heterogeneity Evaluation Approaches for Personalized Federated Learning","date":"2024-10-09","arxiv_id":"2410.07286","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-data-heterogeneity-evaluation#ran","syntology_url":"https://syntology.ai/paper/2410.07286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07286"}},"official":{"repos":["xiaoni-61/dh-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/entering-real-social-world-benchmarking-the","slug":"entering-real-social-world-benchmarking-the","title":"Entering Real Social World! Benchmarking the Social Intelligence of Large Language Models from a First-person Perspective","date":"2024-10-08","arxiv_id":"2410.06195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/entering-real-social-world-benchmarking-the#ran","syntology_url":"https://syntology.ai/paper/2410.06195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.06195"}},"official":{"repos":["gyhou123/egosocialarena"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-glue-democratized-llm-scaling-for-a","slug":"model-glue-democratized-llm-scaling-for-a","title":"Model-GLUE: Democratized LLM Scaling for A Large Model Zoo in the Wild","date":"2024-10-07","arxiv_id":"2410.05357","repositories_listed":1,"syntology":{"n":16,"n_ran":10,"n_constructed":4,"n_ran_checked":6,"n_instrument":4,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/model-glue-democratized-llm-scaling-for-a#ran","syntology_url":"https://syntology.ai/paper/2410.05357","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05357"}},"official":{"repos":["model-glue/model-glue"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":4,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/texttt-dattri-a-library-for-efficient-data","slug":"texttt-dattri-a-library-for-efficient-data","title":"$\\texttt{dattri}$: A Library for Efficient Data Attribution","date":"2024-10-06","arxiv_id":"2410.04555","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/texttt-dattri-a-library-for-efficient-data#ran","syntology_url":"https://syntology.ai/paper/2410.04555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04555"}},"official":{"repos":["trais-lab/dattri"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-large-language-models-for-inverse","slug":"multimodal-large-language-models-for-inverse","title":"Multimodal Large Language Models for Inverse Molecular Design with Retrosynthetic Planning","date":"2024-10-05","arxiv_id":"2410.04223","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multimodal-large-language-models-for-inverse#ran","syntology_url":"https://syntology.ai/paper/2410.04223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04223"}},"official":{"repos":["liugangcode/Llamole"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/autopenbench-benchmarking-generative-agents","slug":"autopenbench-benchmarking-generative-agents","title":"AutoPenBench: Benchmarking Generative Agents for Penetration Testing","date":"2024-10-04","arxiv_id":"2410.03225","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/autopenbench-benchmarking-generative-agents#ran","syntology_url":"https://syntology.ai/paper/2410.03225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03225"}},"official":{"repos":["lucagioacchini/auto-pen-bench","lucagioacchini/genai-pentest-paper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ward-provable-rag-dataset-inference-via-llm","slug":"ward-provable-rag-dataset-inference-via-llm","title":"Ward: Provable RAG Dataset Inference via LLM Watermarks","date":"2024-10-04","arxiv_id":"2410.03537","repositories_listed":0,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ward-provable-rag-dataset-inference-via-llm#ran","syntology_url":"https://syntology.ai/paper/2410.03537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03537"}},"official":null}},{"url":"/paper/how-do-large-language-models-understand-graph","slug":"how-do-large-language-models-understand-graph","title":"How Do Large Language Models Understand Graph Patterns? A Benchmark for Graph Pattern Comprehension","date":"2024-10-04","arxiv_id":"2410.05298","repositories_listed":0,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":14,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-do-large-language-models-understand-graph#ran","syntology_url":"https://syntology.ai/paper/2410.05298","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05298"}},"official":null}},{"url":"/paper/mantra-the-manifold-triangulations-assemblage","slug":"mantra-the-manifold-triangulations-assemblage","title":"MANTRA: The Manifold Triangulations Assemblage","date":"2024-10-03","arxiv_id":"2410.02392","repositories_listed":1,"syntology":{"n":18,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":0,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mantra-the-manifold-triangulations-assemblage#ran","syntology_url":"https://syntology.ai/paper/2410.02392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02392"}},"official":{"repos":["aidos-lab/MANTRA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/agent-security-bench-asb-formalizing-and","slug":"agent-security-bench-asb-formalizing-and","title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","date":"2024-10-03","arxiv_id":"2410.02644","repositories_listed":1,"syntology":{"n":13,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/agent-security-bench-asb-formalizing-and#ran","syntology_url":"https://syntology.ai/paper/2410.02644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02644"}},"official":{"repos":["agiresearch/asb"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/divscene-benchmarking-lvlms-for-object","slug":"divscene-benchmarking-lvlms-for-object","title":"DivScene: Benchmarking LVLMs for Object Navigation with Diverse Scenes and Objects","date":"2024-10-03","arxiv_id":"2410.02730","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divscene-benchmarking-lvlms-for-object#ran","syntology_url":"https://syntology.ai/paper/2410.02730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.02730"}},"official":{"repos":["zhaowei-wang-nlp/divscene"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stringllm-understanding-the-string-processing","slug":"stringllm-understanding-the-string-processing","title":"StringLLM: Understanding the String Processing Capability of Large Language Models","date":"2024-10-02","arxiv_id":"2410.01208","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stringllm-understanding-the-string-processing#ran","syntology_url":"https://syntology.ai/paper/2410.01208","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01208"}},"official":{"repos":["wxl-lxw/stringllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shapiq-shapley-interactions-for-machine","slug":"shapiq-shapley-interactions-for-machine","title":"shapiq: Shapley Interactions for Machine Learning","date":"2024-10-02","arxiv_id":"2410.01649","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shapiq-shapley-interactions-for-machine#ran","syntology_url":"https://syntology.ai/paper/2410.01649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01649"}},"official":{"repos":["mmschlk/shapiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cxpmrg-bench-pre-training-and-benchmarking","slug":"cxpmrg-bench-pre-training-and-benchmarking","title":"CXPMRG-Bench: Pre-training and Benchmarking for X-ray Medical Report Generation on CheXpert Plus Dataset","date":"2024-10-01","arxiv_id":"2410.00379","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cxpmrg-bench-pre-training-and-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2410.00379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00379"}},"official":{"repos":["event-ahu/medical_image_analysis"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-prompts-dynamic-conversational","slug":"beyond-prompts-dynamic-conversational","title":"Beyond Prompts: Dynamic Conversational Benchmarking of Large Language Models","date":"2024-09-30","arxiv_id":"2409.20222","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-prompts-dynamic-conversational#ran","syntology_url":"https://syntology.ai/paper/2409.20222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20222"}},"official":{"repos":["GoodAI/goodai-ltm-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/controlling-risk-of-retrieval-augmented","slug":"controlling-risk-of-retrieval-augmented","title":"Controlling Risk of Retrieval-augmented Generation: A Counterfactual Prompting Framework","date":"2024-09-24","arxiv_id":"2409.16146","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/controlling-risk-of-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2409.16146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16146"}},"official":{"repos":["ict-bigdatalab/rc-rag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rmcbench-benchmarking-large-language-models","slug":"rmcbench-benchmarking-large-language-models","title":"RMCBench: Benchmarking Large Language Models' Resistance to Malicious Code","date":"2024-09-23","arxiv_id":"2409.15154","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rmcbench-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2409.15154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.15154"}},"official":{"repos":["qing-yuan233/RMCBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/paraphrasus-a-comprehensive-benchmark-for","slug":"paraphrasus-a-comprehensive-benchmark-for","title":"PARAPHRASUS : A Comprehensive Benchmark for Evaluating Paraphrase Detection Models","date":"2024-09-18","arxiv_id":"2409.12060","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/paraphrasus-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2409.12060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12060"}},"official":{"repos":["impresso/paraphrasus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thames-an-end-to-end-tool-for-hallucination","slug":"thames-an-end-to-end-tool-for-hallucination","title":"THaMES: An End-to-End Tool for Hallucination Mitigation and Evaluation in Large Language Models","date":"2024-09-17","arxiv_id":"2409.11353","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/thames-an-end-to-end-tool-for-hallucination#ran","syntology_url":"https://syntology.ai/paper/2409.11353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11353"}},"official":{"repos":["holistic-ai/THaMES"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-spurious-bias-in-few-shot-image","slug":"benchmarking-spurious-bias-in-few-shot-image","title":"Benchmarking Spurious Bias in Few-Shot Image Classifiers","date":"2024-09-04","arxiv_id":"2409.02882","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-spurious-bias-in-few-shot-image#ran","syntology_url":"https://syntology.ai/paper/2409.02882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02882"}},"official":{"repos":["gtzheng/fewstab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/spinning-the-golden-thread-benchmarking-long","slug":"spinning-the-golden-thread-benchmarking-long","title":"LongGenBench: Benchmarking Long-Form Generation in Long Context LLMs","date":"2024-09-03","arxiv_id":"2409.02076","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spinning-the-golden-thread-benchmarking-long#ran","syntology_url":"https://syntology.ai/paper/2409.02076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02076"}},"official":{"repos":["mozhu621/SGT","mozhu621/longgenbench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genagent-build-collaborative-ai-systems-with","slug":"genagent-build-collaborative-ai-systems-with","title":"ComfyBench: Benchmarking LLM-based Agents in ComfyUI for Autonomously Designing Collaborative AI Systems","date":"2024-09-02","arxiv_id":"2409.01392","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genagent-build-collaborative-ai-systems-with#ran","syntology_url":"https://syntology.ai/paper/2409.01392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01392"}},"official":{"repos":["xxyQwQ/ComfyBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-agents-simulating-counselor","slug":"interactive-agents-simulating-counselor","title":"Interactive Agents: Simulating Counselor-Client Psychological Counseling via Role-Playing LLM-to-LLM Interactions","date":"2024-08-28","arxiv_id":"2408.15787","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interactive-agents-simulating-counselor#ran","syntology_url":"https://syntology.ai/paper/2408.15787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15787"}},"official":{"repos":["qiuhuachuan/interactive-agents"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wcebleedgen-a-wireless-capsule-endoscopy","slug":"wcebleedgen-a-wireless-capsule-endoscopy","title":"WCEbleedGen: A wireless capsule endoscopy dataset and its benchmarking for automatic bleeding classification, detection, and segmentation","date":"2024-08-22","arxiv_id":"2408.12466","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":17,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/wcebleedgen-a-wireless-capsule-endoscopy#ran","syntology_url":"https://syntology.ai/paper/2408.12466","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12466"}},"official":{"repos":["misahub2023/benchmarking-codes-of-the-wcebleedgen-dataset"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/perturbench-benchmarking-machine-learning","slug":"perturbench-benchmarking-machine-learning","title":"PerturBench: Benchmarking Machine Learning Models for Cellular Perturbation Analysis","date":"2024-08-20","arxiv_id":"2408.10609","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/perturbench-benchmarking-machine-learning#ran","syntology_url":"https://syntology.ai/paper/2408.10609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.10609"}},"official":{"repos":["altoslabs/perturbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/blade-benchmarking-language-model-agents-for","slug":"blade-benchmarking-language-model-agents-for","title":"BLADE: Benchmarking Language Model Agents for Data-Driven Science","date":"2024-08-19","arxiv_id":"2408.09667","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/blade-benchmarking-language-model-agents-for#ran","syntology_url":"https://syntology.ai/paper/2408.09667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09667"}},"official":{"repos":["behavioral-data/blade"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tabularbench-benchmarking-adversarial","slug":"tabularbench-benchmarking-adversarial","title":"TabularBench: Benchmarking Adversarial Robustness for Tabular Deep Learning in Real-world Use-cases","date":"2024-08-14","arxiv_id":"2408.07579","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tabularbench-benchmarking-adversarial#ran","syntology_url":"https://syntology.ai/paper/2408.07579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07579"}},"official":{"repos":["serval-uni-lu/tabularbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sustaindc-benchmarking-for-sustainable-data","slug":"sustaindc-benchmarking-for-sustainable-data","title":"SustainDC: Benchmarking for Sustainable Data Center Control","date":"2024-08-14","arxiv_id":"2408.07841","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sustaindc-benchmarking-for-sustainable-data#ran","syntology_url":"https://syntology.ai/paper/2408.07841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.07841"}},"official":{"repos":["hewlettpackard/dc-rl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dissecting-dissonance-benchmarking-large","slug":"dissecting-dissonance-benchmarking-large","title":"Dissecting Dissonance: Benchmarking Large Multimodal Models Against Self-Contradictory Instructions","date":"2024-08-02","arxiv_id":"2408.01091","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dissecting-dissonance-benchmarking-large#ran","syntology_url":"https://syntology.ai/paper/2408.01091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01091"}},"official":{"repos":["shiyegao/Self-Contradictory-Instructions-SCI"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voxsim-a-perceptual-voice-similarity-dataset","slug":"voxsim-a-perceptual-voice-similarity-dataset","title":"VoxSim: A perceptual voice similarity dataset","date":"2024-07-26","arxiv_id":"2407.18505","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/voxsim-a-perceptual-voice-similarity-dataset#ran","syntology_url":"https://syntology.ai/paper/2407.18505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18505"}},"official":{"repos":["kaistmm/voxsim_trainer"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/is-larger-always-better-evaluating-and","slug":"is-larger-always-better-evaluating-and","title":"ClinicRealm: Re-evaluating Large Language Models with Conventional Machine Learning for Non-Generative Clinical Prediction Tasks","date":"2024-07-26","arxiv_id":"2407.18525","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/is-larger-always-better-evaluating-and#ran","syntology_url":"https://syntology.ai/paper/2407.18525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18525"}},"official":{"repos":["yhzhu99/ehr-llm-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/appworld-a-controllable-world-of-apps-and","slug":"appworld-a-controllable-world-of-apps-and","title":"AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents","date":"2024-07-26","arxiv_id":"2407.18901","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/appworld-a-controllable-world-of-apps-and#ran","syntology_url":"https://syntology.ai/paper/2407.18901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.18901"}},"official":{"repos":["stonybrooknlp/appworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mds-ed-multimodal-decision-support-in-the","slug":"mds-ed-multimodal-decision-support-in-the","title":"Enhancing clinical decision support with physiological waveforms -- a multimodal benchmark in emergency care","date":"2024-07-25","arxiv_id":"2407.17856","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mds-ed-multimodal-decision-support-in-the#ran","syntology_url":"https://syntology.ai/paper/2407.17856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17856"}},"official":{"repos":["ai4healthuol/mds-ed"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/humanvid-demystifying-training-data-for","slug":"humanvid-demystifying-training-data-for","title":"HumanVid: Demystifying Training Data for Camera-controllable Human Image Animation","date":"2024-07-24","arxiv_id":"2407.17438","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/humanvid-demystifying-training-data-for#ran","syntology_url":"https://syntology.ai/paper/2407.17438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17438"}},"official":{"repos":["zhenzhiwang/humanvid"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coala-a-practical-and-vision-centric","slug":"coala-a-practical-and-vision-centric","title":"COALA: A Practical and Vision-Centric Federated Learning Platform","date":"2024-07-23","arxiv_id":"2407.16560","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coala-a-practical-and-vision-centric#ran","syntology_url":"https://syntology.ai/paper/2407.16560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16560"}},"official":{"repos":["sonyresearch/coala"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/abdomenatlas-a-large-scale-detailed-annotated","slug":"abdomenatlas-a-large-scale-detailed-annotated","title":"AbdomenAtlas: A Large-Scale, Detailed-Annotated, & Multi-Center Dataset for Efficient Transfer Learning and Open Algorithmic Benchmarking","date":"2024-07-23","arxiv_id":"2407.16697","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/abdomenatlas-a-large-scale-detailed-annotated#ran","syntology_url":"https://syntology.ai/paper/2407.16697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16697"}},"official":{"repos":["mrgiovanni/abdomenatlas","mrgiovanni/suprem"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lca-on-the-line-benchmarking-out-of","slug":"lca-on-the-line-benchmarking-out-of","title":"LCA-on-the-Line: Benchmarking Out-of-Distribution Generalization with Class Taxonomies","date":"2024-07-22","arxiv_id":"2407.16067","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lca-on-the-line-benchmarking-out-of#ran","syntology_url":"https://syntology.ai/paper/2407.16067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16067"}},"official":{"repos":["elvishelvis/lca-on-the-line"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ecco-can-we-improve-model-generated-code","slug":"ecco-can-we-improve-model-generated-code","title":"ECCO: Can We Improve Model-Generated Code Efficiency Without Sacrificing Functional Correctness?","date":"2024-07-19","arxiv_id":"2407.14044","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ecco-can-we-improve-model-generated-code#ran","syntology_url":"https://syntology.ai/paper/2407.14044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14044"}},"official":{"repos":["codeeff/ecco"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/reliable-and-efficient-concept-erasure-of","slug":"reliable-and-efficient-concept-erasure-of","title":"Reliable and Efficient Concept Erasure of Text-to-Image Diffusion Models","date":"2024-07-17","arxiv_id":"2407.12383","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reliable-and-efficient-concept-erasure-of#ran","syntology_url":"https://syntology.ai/paper/2407.12383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12383"}},"official":{"repos":["charlesgong12/rece"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/abstraction-alignment-comparing-model-and","slug":"abstraction-alignment-comparing-model-and","title":"Abstraction Alignment: Comparing Model-Learned and Human-Encoded Conceptual Relationships","date":"2024-07-17","arxiv_id":"2407.12543","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":16,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/abstraction-alignment-comparing-model-and#ran","syntology_url":"https://syntology.ai/paper/2407.12543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12543"}},"official":{"repos":["mitvis/abstraction-alignment"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-robust-self-supervised-learning","slug":"benchmarking-robust-self-supervised-learning","title":"Benchmarking Robust Self-Supervised Learning Across Diverse Downstream Tasks","date":"2024-07-17","arxiv_id":"2407.12588","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-robust-self-supervised-learning#ran","syntology_url":"https://syntology.ai/paper/2407.12588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12588"}},"official":{"repos":["layer6ai-labs/ssl-robustness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-correctness-benchmarking-multi","slug":"beyond-correctness-benchmarking-multi","title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","date":"2024-07-16","arxiv_id":"2407.11470","repositories_listed":2,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/beyond-correctness-benchmarking-multi#ran","syntology_url":"https://syntology.ai/paper/2407.11470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11470"}},"official":{"repos":["jszheng21/race"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-the-attribution-quality-of","slug":"benchmarking-the-attribution-quality-of","title":"Benchmarking the Attribution Quality of Vision Models","date":"2024-07-16","arxiv_id":"2407.11910","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/benchmarking-the-attribution-quality-of#ran","syntology_url":"https://syntology.ai/paper/2407.11910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11910"}},"official":{"repos":["visinf/idsds"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llms-for-optimization-modeling","slug":"benchmarking-llms-for-optimization-modeling","title":"OptiBench Meets ReSocratic: Measure and Improve LLMs for Optimization Modeling","date":"2024-07-13","arxiv_id":"2407.09887","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-llms-for-optimization-modeling#ran","syntology_url":"https://syntology.ai/paper/2407.09887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09887"}},"official":{"repos":["yangzhch6/ReSocratic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-language-model-creativity-a-case","slug":"benchmarking-language-model-creativity-a-case","title":"Benchmarking Language Model Creativity: A Case Study on Code Generation","date":"2024-07-12","arxiv_id":"2407.09007","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-language-model-creativity-a-case#ran","syntology_url":"https://syntology.ai/paper/2407.09007","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09007"}},"official":{"repos":["JHU-CLSP/NeoCoder"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/retrospective-for-the-dynamic-sensorium","slug":"retrospective-for-the-dynamic-sensorium","title":"Retrospective for the Dynamic Sensorium Competition for predicting large-scale mouse primary visual cortex activity from videos","date":"2024-07-12","arxiv_id":"2407.09100","repositories_listed":2,"syntology":{"n":21,"n_ran":17,"n_constructed":0,"n_ran_checked":16,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":11,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/retrospective-for-the-dynamic-sensorium#ran","syntology_url":"https://syntology.ai/paper/2407.09100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09100"}},"official":{"repos":["ecker-lab/sensorium_2023","bryanlimy/ViV1T"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/wayvescenes101-a-dataset-and-benchmark-for","slug":"wayvescenes101-a-dataset-and-benchmark-for","title":"WayveScenes101: A Dataset and Benchmark for Novel View Synthesis in Autonomous Driving","date":"2024-07-11","arxiv_id":"2407.08280","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/wayvescenes101-a-dataset-and-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2407.08280","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08280"}},"official":{"repos":["wayveai/wayve_scenes"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/natural-language-is-not-enough-benchmarking","slug":"natural-language-is-not-enough-benchmarking","title":"Natural language is not enough: Benchmarking multi-modal generative AI for Verilog generation","date":"2024-07-11","arxiv_id":"2407.08473","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/natural-language-is-not-enough-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2407.08473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08473"}},"official":{"repos":["aichipdesign/chipgptv"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-embedding-aggregation-methods-in","slug":"benchmarking-embedding-aggregation-methods-in","title":"Benchmarking Embedding Aggregation Methods in Computational Pathology: A Clinical Data Perspective","date":"2024-07-10","arxiv_id":"2407.07841","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-embedding-aggregation-methods-in#ran","syntology_url":"https://syntology.ai/paper/2407.07841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07841"}},"official":{"repos":["fuchs-lab-public/cpath_sabenchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/training-on-the-test-task-confounds","slug":"training-on-the-test-task-confounds","title":"Training on the Test Task Confounds Evaluation and Emergence","date":"2024-07-10","arxiv_id":"2407.07890","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/training-on-the-test-task-confounds#ran","syntology_url":"https://syntology.ai/paper/2407.07890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07890"}},"official":{"repos":["socialfoundations/training-on-the-test-task"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-benchmarking-and-understanding-1","slug":"revisiting-benchmarking-and-understanding-1","title":"Revisiting, Benchmarking and Understanding Unsupervised Graph Domain Adaptation","date":"2024-07-09","arxiv_id":"2407.11052","repositories_listed":1,"syntology":{"n":47,"n_ran":38,"n_constructed":0,"n_ran_checked":32,"n_instrument":6,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":31,"n_pointer_only":21,"phrase":"38 ran (of which 0 constructed an object rather than computing a result; 32 with no instrument failure: 1 honoured, 0 violated, 31 with no contract checked; 6 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/revisiting-benchmarking-and-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2407.11052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11052"}},"official":{"repos":["pygda-team/pygda"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/simulation-based-benchmarking-for-causal","slug":"simulation-based-benchmarking-for-causal","title":"Simulation-based Benchmarking for Causal Structure Learning in Gene Perturbation Experiments","date":"2024-07-08","arxiv_id":"2407.06015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simulation-based-benchmarking-for-causal#ran","syntology_url":"https://syntology.ai/paper/2407.06015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06015"}},"official":{"repos":["luka-kovacevic/causalregnet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-the-effectiveness-of-graph","slug":"rethinking-the-effectiveness-of-graph","title":"Rethinking the Effectiveness of Graph Classification Datasets in Benchmarks for Assessing GNNs","date":"2024-07-06","arxiv_id":"2407.04999","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/rethinking-the-effectiveness-of-graph#ran","syntology_url":"https://syntology.ai/paper/2407.04999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04999"}},"official":{"repos":["ICLab4DL/GNNBenchEffectiveness"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"0c56e7d42cd28903eb854b550ab487483909dc8d094d8bef6f68259114868d8c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}