{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/ran/1","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":8,"rows_per_page":100,"rows":[1,100],"of":749,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking/papers/ran/1","prev":null,"next":"/task/benchmarking/papers/ran/2","papers":[{"url":"/paper/drafterbench-benchmarking-large-language","slug":"drafterbench-benchmarking-large-language","title":"DrafterBench: Benchmarking Large Language Models for Tasks Automation in Civil Engineering","date":"2025-07-15","arxiv_id":"2507.11527","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/drafterbench-benchmarking-large-language#ran","syntology_url":"https://syntology.ai/paper/2507.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11527"}},"official":{"repos":["eason-li-ais/drafterbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/codeassistbench-cab-dataset-benchmarking-for","slug":"codeassistbench-cab-dataset-benchmarking-for","title":"CodeAssistBench (CAB): Dataset & Benchmarking for Multi-turn Chat-Based Code Assistance","date":"2025-07-14","arxiv_id":"2507.10646","repositories_listed":0,"syntology":{"n":24,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/codeassistbench-cab-dataset-benchmarking-for#ran","syntology_url":"https://syntology.ai/paper/2507.10646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.10646"}},"official":null}},{"url":"/paper/multihuman-testbench-benchmarking-image","slug":"multihuman-testbench-benchmarking-image","title":"MultiHuman-Testbench: Benchmarking Image Generation for Multiple Humans","date":"2025-06-25","arxiv_id":"2506.20879","repositories_listed":0,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multihuman-testbench-benchmarking-image#ran","syntology_url":"https://syntology.ai/paper/2506.20879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20879"}},"official":null}},{"url":"/paper/tab-unified-benchmarking-of-time-series","slug":"tab-unified-benchmarking-of-time-series","title":"TAB: Unified Benchmarking of Time Series Anomaly Detection Methods","date":"2025-06-22","arxiv_id":"2506.18046","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tab-unified-benchmarking-of-time-series#ran","syntology_url":"https://syntology.ai/paper/2506.18046","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18046"}},"official":{"repos":["decisionintelligence/tab"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tabarena-a-living-benchmark-for-machine","slug":"tabarena-a-living-benchmark-for-machine","title":"TabArena: A Living Benchmark for Machine Learning on Tabular Data","date":"2025-06-20","arxiv_id":"2506.16791","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tabarena-a-living-benchmark-for-machine#ran","syntology_url":"https://syntology.ai/paper/2506.16791","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.16791"}},"official":null}},{"url":"/paper/impliret-benchmarking-the-implicit-fact","slug":"impliret-benchmarking-the-implicit-fact","title":"ImpliRet: Benchmarking the Implicit Fact Retrieval Challenge","date":"2025-06-17","arxiv_id":"2506.14407","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/impliret-benchmarking-the-implicit-fact#ran","syntology_url":"https://syntology.ai/paper/2506.14407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.14407"}},"official":{"repos":["zeinabtaghavi/impliret"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-price-of-freedom-exploring-expressivity","slug":"the-price-of-freedom-exploring-expressivity","title":"The Price of Freedom: Exploring Expressivity and Runtime Tradeoffs in Equivariant Tensor Products","date":"2025-06-16","arxiv_id":"2506.13523","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-price-of-freedom-exploring-expressivity#ran","syntology_url":"https://syntology.ai/paper/2506.13523","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13523"}},"official":{"repos":["atomicarchitects/priceoffreedom"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openunlearning-accelerating-llm-unlearning","slug":"openunlearning-accelerating-llm-unlearning","title":"OpenUnlearning: Accelerating LLM Unlearning via Unified Benchmarking of Methods and Metrics","date":"2025-06-14","arxiv_id":"2506.12618","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/openunlearning-accelerating-llm-unlearning#ran","syntology_url":"https://syntology.ai/paper/2506.12618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.12618"}},"official":{"repos":["locuslab/open-unlearning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/glgenn-a-novel-parameter-light-equivariant","slug":"glgenn-a-novel-parameter-light-equivariant","title":"GLGENN: A Novel Parameter-Light Equivariant Neural Networks Architecture Based on Clifford Geometric Algebras","date":"2025-06-11","arxiv_id":"2506.09625","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/glgenn-a-novel-parameter-light-equivariant#ran","syntology_url":"https://syntology.ai/paper/2506.09625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09625"}},"official":{"repos":["katyafilimoshina/glgenn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hopadiff-holistic-partial-aware-fourier","slug":"hopadiff-holistic-partial-aware-fourier","title":"HopaDIFF: Holistic-Partial Aware Fourier Conditioned Diffusion for Referring Human Action Segmentation in Multi-Person Scenarios","date":"2025-06-11","arxiv_id":"2506.09650","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hopadiff-holistic-partial-aware-fourier#ran","syntology_url":"https://syntology.ai/paper/2506.09650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.09650"}},"official":{"repos":["kpeng9510/hopadiff"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-catechol-benchmark-time-series-solvent","slug":"the-catechol-benchmark-time-series-solvent","title":"The Catechol Benchmark: Time-series Solvent Selection Data for Few-shot Machine Learning","date":"2025-06-09","arxiv_id":"2506.07619","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-catechol-benchmark-time-series-solvent#ran","syntology_url":"https://syntology.ai/paper/2506.07619","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07619"}},"official":{"repos":["jpfolch/catechol_solvent_selection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2506-08249","slug":"2506-08249","title":"RADAR: Benchmarking Language Models on Imperfect Tabular Data","date":"2025-06-09","arxiv_id":"2506.08249","repositories_listed":1,"syntology":{"n":15,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2506-08249#ran","syntology_url":"https://syntology.ai/paper/2506.08249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08249"}},"official":{"repos":["kenqgu/radar"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mca-bench-a-multimodal-benchmark-for","slug":"mca-bench-a-multimodal-benchmark-for","title":"MCA-Bench: A Multimodal Benchmark for Evaluating CAPTCHA Robustness Against VLM-based Attacks","date":"2025-06-06","arxiv_id":"2506.05982","repositories_listed":2,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mca-bench-a-multimodal-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2506.05982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05982"}},"official":{"repos":["noheadwuzonglin/mca-bench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/debatable-intelligence-benchmarking-llm","slug":"debatable-intelligence-benchmarking-llm","title":"Debatable Intelligence: Benchmarking LLM Judges via Debate Speech Evaluation","date":"2025-06-05","arxiv_id":"2506.05062","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/debatable-intelligence-benchmarking-llm#ran","syntology_url":"https://syntology.ai/paper/2506.05062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.05062"}},"official":{"repos":["noy-sternlicht/debatable-intelligence"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-machine-unlearning-in-image","slug":"rethinking-machine-unlearning-in-image","title":"Rethinking Machine Unlearning in Image Generation Models","date":"2025-06-03","arxiv_id":"2506.02761","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rethinking-machine-unlearning-in-image#ran","syntology_url":"https://syntology.ai/paper/2506.02761","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.02761"}},"official":{"repos":["ryliu68/igmu"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/texttt-avrobustbench-benchmarking-the","slug":"texttt-avrobustbench-benchmarking-the","title":"$\\texttt{AVROBUSTBENCH}$: Benchmarking the Robustness of Audio-Visual Recognition Models at Test-Time","date":"2025-05-31","arxiv_id":"2506.00358","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/texttt-avrobustbench-benchmarking-the#ran","syntology_url":"https://syntology.ai/paper/2506.00358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00358"}},"official":{"repos":["sarthaxxxxx/av-c-robustness-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/draw-all-your-imagine-a-holistic-benchmark","slug":"draw-all-your-imagine-a-holistic-benchmark","title":"Draw ALL Your Imagine: A Holistic Benchmark and Agent Framework for Complex Instruction-based Image Generation","date":"2025-05-30","arxiv_id":"2505.24787","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/draw-all-your-imagine-a-holistic-benchmark#ran","syntology_url":"https://syntology.ai/paper/2505.24787","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24787"}},"official":{"repos":["yczhou001/longbench-t2i"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-atomic-geometry-representations-in","slug":"beyond-atomic-geometry-representations-in","title":"Beyond Atomic Geometry Representations in Materials Science: A Human-in-the-Loop Multimodal Framework","date":"2025-05-30","arxiv_id":"2506.00302","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/beyond-atomic-geometry-representations-in#ran","syntology_url":"https://syntology.ai/paper/2506.00302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.00302"}},"official":{"repos":["kurbanintelligencelab/multicrystalspectrumset"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/svrpbench-a-realistic-benchmark-for","slug":"svrpbench-a-realistic-benchmark-for","title":"SVRPBench: A Realistic Benchmark for Stochastic Vehicle Routing Problem","date":"2025-05-28","arxiv_id":"2505.21887","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/svrpbench-a-realistic-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2505.21887","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21887"}},"official":{"repos":["yehias21/vrp-benchmarks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/redteamcua-realistic-adversarial-testing-of","slug":"redteamcua-realistic-adversarial-testing-of","title":"RedTeamCUA: Realistic Adversarial Testing of Computer-Use Agents in Hybrid Web-OS Environments","date":"2025-05-28","arxiv_id":"2505.21936","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/redteamcua-realistic-adversarial-testing-of#ran","syntology_url":"https://syntology.ai/paper/2505.21936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21936"}},"official":{"repos":["osu-nlp-group/redteamcua"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/videomarkbench-benchmarking-robustness-of","slug":"videomarkbench-benchmarking-robustness-of","title":"VideoMarkBench: Benchmarking Robustness of Video Watermarking","date":"2025-05-27","arxiv_id":"2505.21620","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomarkbench-benchmarking-robustness-of#ran","syntology_url":"https://syntology.ai/paper/2505.21620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.21620"}},"official":{"repos":["zhengyuan-jiang/videomarkbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/finlora-benchmarking-lora-methods-for-fine","slug":"finlora-benchmarking-lora-methods-for-fine","title":"FinLoRA: Benchmarking LoRA Methods for Fine-Tuning LLMs on Financial Datasets","date":"2025-05-26","arxiv_id":"2505.19819","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finlora-benchmarking-lora-methods-for-fine#ran","syntology_url":"https://syntology.ai/paper/2505.19819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19819"}},"official":{"repos":["open-finance-lab/finlora"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mineanybuild-benchmarking-spatial-planning","slug":"mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","arxiv_id":"2505.20148","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineanybuild-benchmarking-spatial-planning#ran","syntology_url":"https://syntology.ai/paper/2505.20148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20148"}},"official":{"repos":["mineanybuild/mineanybuild"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/are-vision-language-models-ready-for-clinical","slug":"are-vision-language-models-ready-for-clinical","title":"Are Vision Language Models Ready for Clinical Diagnosis? A 3D Medical Benchmark for Tumor-centric Visual Question Answering","date":"2025-05-25","arxiv_id":"2505.18915","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vision-language-models-ready-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2505.18915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18915"}},"official":{"repos":["schuture/deeptumorvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chartgalaxy-a-dataset-for-infographic-chart","slug":"chartgalaxy-a-dataset-for-infographic-chart","title":"ChartGalaxy: A Dataset for Infographic Chart Understanding and Generation","date":"2025-05-24","arxiv_id":"2505.18668","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartgalaxy-a-dataset-for-infographic-chart#ran","syntology_url":"https://syntology.ai/paper/2505.18668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18668"}},"official":{"repos":["chartgalaxy/chartgalaxy"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/twin-2k-500-a-dataset-for-building-digital","slug":"twin-2k-500-a-dataset-for-building-digital","title":"Twin-2K-500: A dataset for building digital twins of over 2,000 people based on their answers to over 500 questions","date":"2025-05-23","arxiv_id":"2505.17479","repositories_listed":1,"syntology":{"n":16,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/twin-2k-500-a-dataset-for-building-digital#ran","syntology_url":"https://syntology.ai/paper/2505.17479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17479"}},"official":{"repos":["tianyipeng-lab/digital-twin-simulation"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/audiotrust-benchmarking-the-multifaceted","slug":"audiotrust-benchmarking-the-multifaceted","title":"AudioTrust: Benchmarking the Multifaceted Trustworthiness of Audio Large Language Models","date":"2025-05-22","arxiv_id":"2505.16211","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiotrust-benchmarking-the-multifaceted#ran","syntology_url":"https://syntology.ai/paper/2505.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16211"}},"official":{"repos":["jusperlee/audiotrust"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-retrieval-augmented-multimomal","slug":"benchmarking-retrieval-augmented-multimomal","title":"Benchmarking Retrieval-Augmented Multimomal Generation for Document Question Answering","date":"2025-05-22","arxiv_id":"2505.16470","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-multimomal#ran","syntology_url":"https://syntology.ai/paper/2505.16470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16470"}},"official":{"repos":["mmdocrag/mmdocrag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tropical-attention-neural-algorithmic","slug":"tropical-attention-neural-algorithmic","title":"Tropical Attention: Neural Algorithmic Reasoning for Combinatorial Algorithms","date":"2025-05-22","arxiv_id":"2505.17190","repositories_listed":0,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tropical-attention-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2505.17190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17190"}},"official":null}},{"url":"/paper/lost-in-benchmarks-rethinking-large-language","slug":"lost-in-benchmarks-rethinking-large-language","title":"Lost in Benchmarks? Rethinking Large Language Model Benchmarking with Item Response Theory","date":"2025-05-21","arxiv_id":"2505.15055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lost-in-benchmarks-rethinking-large-language#ran","syntology_url":"https://syntology.ai/paper/2505.15055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.15055"}},"official":{"repos":["Joe-Hall-Lee/PSN-IRT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ineq-comp-benchmarking-human-intuitive","slug":"ineq-comp-benchmarking-human-intuitive","title":"Ineq-Comp: Benchmarking Human-Intuitive Compositional Reasoning in Automated Theorem Proving on Inequalities","date":"2025-05-19","arxiv_id":"2505.12680","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ineq-comp-benchmarking-human-intuitive#ran","syntology_url":"https://syntology.ai/paper/2505.12680","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12680"}},"official":{"repos":["haoyuzhao123/leanineqcomp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medagentboard-benchmarking-multi-agent","slug":"medagentboard-benchmarking-multi-agent","title":"MedAgentBoard: Benchmarking Multi-Agent Collaboration with Conventional Methods for Diverse Medical Tasks","date":"2025-05-18","arxiv_id":"2505.12371","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentboard-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.12371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12371"}},"official":{"repos":["yhzhu99/medagentboard"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/time-travel-is-cheating-going-live-with","slug":"time-travel-is-cheating-going-live-with","title":"Time Travel is Cheating: Going Live with DeepFund for Real-Time Fund Investment Benchmarking","date":"2025-05-16","arxiv_id":"2505.11065","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/time-travel-is-cheating-going-live-with#ran","syntology_url":"https://syntology.ai/paper/2505.11065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11065"}},"official":{"repos":["hkustdial/deepfund"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-10610","slug":"2505-10610","title":"MMLongBench: Benchmarking Long-Context Vision-Language Models Effectively and Thoroughly","date":"2025-05-15","arxiv_id":"2505.10610","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10610#ran","syntology_url":"https://syntology.ai/paper/2505.10610","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10610"}},"official":{"repos":["edinburghnlp/mmlongbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-ml-energy-benchmark-toward-automated","slug":"the-ml-energy-benchmark-toward-automated","title":"The ML.ENERGY Benchmark: Toward Automated Inference Energy Measurement and Optimization","date":"2025-05-09","arxiv_id":"2505.06371","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-ml-energy-benchmark-toward-automated#ran","syntology_url":"https://syntology.ai/paper/2505.06371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.06371"}},"official":{"repos":["ml-energy/leaderboard"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/enhancing-treatment-effect-estimation-via","slug":"enhancing-treatment-effect-estimation-via","title":"Enhancing Treatment Effect Estimation via Active Learning: A Counterfactual Covering Perspective","date":"2025-05-08","arxiv_id":"2505.05242","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-treatment-effect-estimation-via#ran","syntology_url":"https://syntology.ai/paper/2505.05242","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05242"}},"official":{"repos":["uqhwen2/FCCM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llm-faithfulness-in-rag-with","slug":"benchmarking-llm-faithfulness-in-rag-with","title":"Benchmarking LLM Faithfulness in RAG with Evolving Leaderboards","date":"2025-05-07","arxiv_id":"2505.04847","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/benchmarking-llm-faithfulness-in-rag-with#ran","syntology_url":"https://syntology.ai/paper/2505.04847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04847"}},"official":{"repos":["vectara/FaithJudge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/combibench-benchmarking-llm-capability-for","slug":"combibench-benchmarking-llm-capability-for","title":"CombiBench: Benchmarking LLM Capability for Combinatorial Mathematics","date":"2025-05-06","arxiv_id":"2505.03171","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/combibench-benchmarking-llm-capability-for#ran","syntology_url":"https://syntology.ai/paper/2505.03171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.03171"}},"official":{"repos":["moonshotai/combibench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/formalmath-benchmarking-formal-mathematical","slug":"formalmath-benchmarking-formal-mathematical","title":"FormalMATH: Benchmarking Formal Mathematical Reasoning of Large Language Models","date":"2025-05-05","arxiv_id":"2505.02735","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/formalmath-benchmarking-formal-mathematical#ran","syntology_url":"https://syntology.ai/paper/2505.02735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02735"}},"official":{"repos":["sphere-ai-lab/formalmath-bench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/physics-learning-ai-datamodel-plaid-datasets","slug":"physics-learning-ai-datamodel-plaid-datasets","title":"Physics-Learning AI Datamodel (PLAID) datasets: a collection of physics simulations for machine learning","date":"2025-05-05","arxiv_id":"2505.02974","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/physics-learning-ai-datamodel-plaid-datasets#ran","syntology_url":"https://syntology.ai/paper/2505.02974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02974"}},"official":null}},{"url":"/paper/meta-black-box-optimization-through-offline-q","slug":"meta-black-box-optimization-through-offline-q","title":"Meta-Black-Box-Optimization through Offline Q-function Learning","date":"2025-05-04","arxiv_id":"2505.02010","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/meta-black-box-optimization-through-offline-q#ran","syntology_url":"https://syntology.ai/paper/2505.02010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02010"}},"official":{"repos":["metaevo/q-mamba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/rtv-bench-benchmarking-mllm-continuous","slug":"rtv-bench-benchmarking-mllm-continuous","title":"RTV-Bench: Benchmarking MLLM Continuous Perception, Understanding and Reasoning through Real-Time Video","date":"2025-05-04","arxiv_id":"2505.02064","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rtv-bench-benchmarking-mllm-continuous#ran","syntology_url":"https://syntology.ai/paper/2505.02064","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.02064"}},"official":{"repos":["ljungang/rtv-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/osvbench-benchmarking-llms-on-specification","slug":"osvbench-benchmarking-llms-on-specification","title":"OSVBench: Benchmarking LLMs on Specification Generation Tasks for Operating System Verification","date":"2025-04-29","arxiv_id":"2504.20964","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/osvbench-benchmarking-llms-on-specification#ran","syntology_url":"https://syntology.ai/paper/2504.20964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20964"}},"official":{"repos":["lishangyu-hkust/osvbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/browsecomp-zh-benchmarking-web-browsing","slug":"browsecomp-zh-benchmarking-web-browsing","title":"BrowseComp-ZH: Benchmarking Web Browsing Ability of Large Language Models in Chinese","date":"2025-04-27","arxiv_id":"2504.19314","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/browsecomp-zh-benchmarking-web-browsing#ran","syntology_url":"https://syntology.ai/paper/2504.19314","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.19314"}},"official":{"repos":["palin2018/browsecomp-zh"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-vision-language-models-on","slug":"benchmarking-large-vision-language-models-on","title":"Benchmarking Large Vision-Language Models on Fine-Grained Image Tasks: A Comprehensive Evaluation","date":"2025-04-21","arxiv_id":"2504.14988","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-vision-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2504.14988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14988"}},"official":{"repos":["seu-vipgroup/fg-bmk"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/know-me-respond-to-me-benchmarking-llms-for","slug":"know-me-respond-to-me-benchmarking-llms-for","title":"Know Me, Respond to Me: Benchmarking LLMs for Dynamic User Profiling and Personalized Responses at Scale","date":"2025-04-19","arxiv_id":"2504.14225","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/know-me-respond-to-me-benchmarking-llms-for#ran","syntology_url":"https://syntology.ai/paper/2504.14225","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.14225"}},"official":{"repos":["bowen-upenn/personamem"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fhbench-towards-efficient-and-personalized","slug":"fhbench-towards-efficient-and-personalized","title":"FHBench: Towards Efficient and Personalized Federated Learning for Multimodal Healthcare","date":"2025-04-15","arxiv_id":"2504.10817","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fhbench-towards-efficient-and-personalized#ran","syntology_url":"https://syntology.ai/paper/2504.10817","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10817"}},"official":{"repos":["wph6/fhbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hypobench-towards-systematic-and-principled","slug":"hypobench-towards-systematic-and-principled","title":"HypoBench: Towards Systematic and Principled Benchmarking for Hypothesis Generation","date":"2025-04-15","arxiv_id":"2504.11524","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypobench-towards-systematic-and-principled#ran","syntology_url":"https://syntology.ai/paper/2504.11524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11524"}},"official":null}},{"url":"/paper/real-benchmarking-autonomous-agents-on","slug":"real-benchmarking-autonomous-agents-on","title":"REAL: Benchmarking Autonomous Agents on Deterministic Simulations of Real Websites","date":"2025-04-15","arxiv_id":"2504.11543","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-benchmarking-autonomous-agents-on#ran","syntology_url":"https://syntology.ai/paper/2504.11543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.11543"}},"official":{"repos":["agi-inc/real","agi-inc/agisdk"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-biopharmaceuticals-retrieval","slug":"benchmarking-biopharmaceuticals-retrieval","title":"Benchmarking Biopharmaceuticals Retrieval-Augmented Generation Evaluation","date":"2025-04-15","arxiv_id":"2504.12342","repositories_listed":0,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-biopharmaceuticals-retrieval#ran","syntology_url":"https://syntology.ai/paper/2504.12342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.12342"}},"official":null}},{"url":"/paper/are-you-getting-what-you-pay-for-auditing","slug":"are-you-getting-what-you-pay-for-auditing","title":"Are You Getting What You Pay For? Auditing Model Substitution in LLM APIs","date":"2025-04-07","arxiv_id":"2504.04715","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/are-you-getting-what-you-pay-for-auditing#ran","syntology_url":"https://syntology.ai/paper/2504.04715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04715"}},"official":{"repos":["sunblaze-ucb/llm-api-audit"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/co-bench-benchmarking-language-model-agents","slug":"co-bench-benchmarking-language-model-agents","title":"CO-Bench: Benchmarking Language Model Agents in Algorithm Search for Combinatorial Optimization","date":"2025-04-06","arxiv_id":"2504.04310","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-bench-benchmarking-language-model-agents#ran","syntology_url":"https://syntology.ai/paper/2504.04310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04310"}},"official":{"repos":["sunnweiwei/co-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/envisioning-beyond-the-pixels-benchmarking","slug":"envisioning-beyond-the-pixels-benchmarking","title":"Envisioning Beyond the Pixels: Benchmarking Reasoning-Informed Visual Editing","date":"2025-04-03","arxiv_id":"2504.02826","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/envisioning-beyond-the-pixels-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2504.02826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.02826"}},"official":{"repos":["phoenixz810/risebench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-synthetic-tabular-data-a-multi","slug":"benchmarking-synthetic-tabular-data-a-multi","title":"Benchmarking Synthetic Tabular Data: A Multi-Dimensional Evaluation Framework","date":"2025-04-02","arxiv_id":"2504.01908","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-synthetic-tabular-data-a-multi#ran","syntology_url":"https://syntology.ai/paper/2504.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.01908"}},"official":{"repos":["mostly-ai/mostlyai-qa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/facebench-a-multi-view-multi-level-facial","slug":"facebench-a-multi-view-multi-level-facial","title":"FaceBench: A Multi-View Multi-Level Facial Attribute VQA Dataset for Benchmarking Face Perception MLLMs","date":"2025-03-27","arxiv_id":"2503.21457","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/facebench-a-multi-view-multi-level-facial#ran","syntology_url":"https://syntology.ai/paper/2503.21457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21457"}},"official":{"repos":["cvi-szu/facebench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-optimizing-organism-wide","slug":"benchmarking-and-optimizing-organism-wide","title":"Benchmarking and optimizing organism wide single-cell RNA alignment methods","date":"2025-03-26","arxiv_id":"2503.20730","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-and-optimizing-organism-wide#ran","syntology_url":"https://syntology.ai/paper/2503.20730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20730"}},"official":{"repos":["phenomicai/bascvi"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-gap-benchmarking-spatial-reasoning","slug":"mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19707","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-gap-benchmarking-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19707"}},"official":{"repos":["stogiannidis/srbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accurate-peak-detection-in-multimodal","slug":"accurate-peak-detection-in-multimodal","title":"Accurate Peak Detection in Multimodal Optimization via Approximated Landscape Learning","date":"2025-03-23","arxiv_id":"2503.18066","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/accurate-peak-detection-in-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.18066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.18066"}},"official":{"repos":["gmc-drl/apdmmo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/nninteractive-redefining-3d-promptable","slug":"nninteractive-redefining-3d-promptable","title":"nnInteractive: Redefining 3D Promptable Segmentation","date":"2025-03-11","arxiv_id":"2503.08373","repositories_listed":4,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/nninteractive-redefining-3d-promptable#ran","syntology_url":"https://syntology.ai/paper/2503.08373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08373"}},"official":{"repos":["mic-dkfz/napari-nninteractive","mic-dkfz/nninteractive"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/medagentsbench-benchmarking-thinking-models","slug":"medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","arxiv_id":"2503.07459","repositories_listed":1,"syntology":{"n":23,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":4,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentsbench-benchmarking-thinking-models#ran","syntology_url":"https://syntology.ai/paper/2503.07459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07459"}},"official":{"repos":["gersteinlab/medagents-benchmark"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/2503-00089","slug":"2503-00089","title":"Protein Structure Tokenization: Benchmarking and New Recipe","date":"2025-02-28","arxiv_id":"2503.00089","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/2503-00089#ran","syntology_url":"https://syntology.ai/paper/2503.00089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00089"}},"official":{"repos":["katarinayuan/structtokenbench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/collab-overcooked-benchmarking-and-evaluating","slug":"collab-overcooked-benchmarking-and-evaluating","title":"Collab-Overcooked: Benchmarking and Evaluating Large Language Models as Collaborative Agents","date":"2025-02-27","arxiv_id":"2502.20073","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collab-overcooked-benchmarking-and-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.20073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20073"}},"official":{"repos":["yusaemeow/collab-overcooked"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/egonormia-benchmarking-physical-social-norm","slug":"egonormia-benchmarking-physical-social-norm","title":"EgoNormia: Benchmarking Physical Social Norm Understanding","date":"2025-02-27","arxiv_id":"2502.20490","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/egonormia-benchmarking-physical-social-norm#ran","syntology_url":"https://syntology.ai/paper/2502.20490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20490"}},"official":{"repos":["open-social-world/egonormia"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-graph-tasks-with-pure-llms-a","slug":"exploring-graph-tasks-with-pure-llms-a","title":"Exploring Graph Tasks with Pure LLMs: A Comprehensive Benchmark and Investigation","date":"2025-02-26","arxiv_id":"2502.18771","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/exploring-graph-tasks-with-pure-llms-a#ran","syntology_url":"https://syntology.ai/paper/2502.18771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18771"}},"official":{"repos":["myflashbarry/LLM-benchmarking"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/an-analyst-inspector-framework-for-evaluating","slug":"an-analyst-inspector-framework-for-evaluating","title":"An Analyst-Inspector Framework for Evaluating Reproducibility of LLMs in Data Science","date":"2025-02-23","arxiv_id":"2502.16395","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-analyst-inspector-framework-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.16395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16395"}},"official":{"repos":["qunhualilab/llm-ds-reproducibility"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adversarial-prompt-evaluation-systematic","slug":"adversarial-prompt-evaluation-systematic","title":"Adversarial Prompt Evaluation: Systematic Benchmarking of Guardrails Against Prompt Input Attacks on LLMs","date":"2025-02-21","arxiv_id":"2502.15427","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-prompt-evaluation-systematic#ran","syntology_url":"https://syntology.ai/paper/2502.15427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.15427"}},"official":{"repos":["ibm/adversarial-prompt-evaluation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ilias-instance-level-image-retrieval-at-scale","slug":"ilias-instance-level-image-retrieval-at-scale","title":"ILIAS: Instance-Level Image retrieval At Scale","date":"2025-02-17","arxiv_id":"2502.11748","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ilias-instance-level-image-retrieval-at-scale#ran","syntology_url":"https://syntology.ai/paper/2502.11748","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11748"}},"official":null}},{"url":"/paper/do-llms-recognize-your-preferences-evaluating","slug":"do-llms-recognize-your-preferences-evaluating","title":"Do LLMs Recognize Your Preferences? Evaluating Personalized Preference Following in LLMs","date":"2025-02-13","arxiv_id":"2502.09597","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/do-llms-recognize-your-preferences-evaluating#ran","syntology_url":"https://syntology.ai/paper/2502.09597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09597"}},"official":{"repos":["amazon-science/PrefEval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fino1-on-the-transferability-of-reasoning","slug":"fino1-on-the-transferability-of-reasoning","title":"Fino1: On the Transferability of Reasoning Enhanced LLMs to Finance","date":"2025-02-12","arxiv_id":"2502.08127","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fino1-on-the-transferability-of-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.08127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.08127"}},"official":{"repos":["the-finai/fino1"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-vision-language-models-on","slug":"benchmarking-vision-language-models-on","title":"Benchmarking Vision-Language Models on Optical Character Recognition in Dynamic Video Environments","date":"2025-02-10","arxiv_id":"2502.06445","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-vision-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2502.06445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06445"}},"official":{"repos":["video-db/ocr-benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mol-moe-training-preference-guided-routers","slug":"mol-moe-training-preference-guided-routers","title":"Mol-MoE: Training Preference-Guided Routers for Molecule Generation","date":"2025-02-08","arxiv_id":"2502.05633","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mol-moe-training-preference-guided-routers#ran","syntology_url":"https://syntology.ai/paper/2502.05633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05633"}},"official":{"repos":["ddidacus/mol-moe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-extended-benchmarking-of-multi-agent","slug":"an-extended-benchmarking-of-multi-agent","title":"An Extended Benchmarking of Multi-Agent Reinforcement Learning Algorithms in Complex Fully Cooperative Tasks","date":"2025-02-07","arxiv_id":"2502.04773","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-extended-benchmarking-of-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2502.04773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04773"}},"official":{"repos":["ailabdsunipi/pymarlzooplus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/itbench-evaluating-ai-agents-across-diverse","slug":"itbench-evaluating-ai-agents-across-diverse","title":"ITBench: Evaluating AI Agents across Diverse Real-World IT Automation Tasks","date":"2025-02-07","arxiv_id":"2502.05352","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/itbench-evaluating-ai-agents-across-diverse#ran","syntology_url":"https://syntology.ai/paper/2502.05352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05352"}},"official":{"repos":["IBM/itbench-sample-scenarios","IBM/itbench-sre-agent"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/speculative-prefill-turbocharging-ttft-with","slug":"speculative-prefill-turbocharging-ttft-with","title":"Speculative Prefill: Turbocharging TTFT with Lightweight and Training-Free Token Importance Estimation","date":"2025-02-05","arxiv_id":"2502.02789","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/speculative-prefill-turbocharging-ttft-with#ran","syntology_url":"https://syntology.ai/paper/2502.02789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02789"}},"official":{"repos":["Jingyu6/speculative_prefill"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tgb-seq-benchmark-challenging-temporal-gnns","slug":"tgb-seq-benchmark-challenging-temporal-gnns","title":"TGB-Seq Benchmark: Challenging Temporal GNNs with Complex Sequential Dynamics","date":"2025-02-05","arxiv_id":"2502.02975","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tgb-seq-benchmark-challenging-temporal-gnns#ran","syntology_url":"https://syntology.ai/paper/2502.02975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02975"}},"official":{"repos":["TGB-Seq/TGB-Seq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/no-metric-to-rule-them-all-toward-principled","slug":"no-metric-to-rule-them-all-toward-principled","title":"No Metric to Rule Them All: Toward Principled Evaluations of Graph-Learning Datasets","date":"2025-02-04","arxiv_id":"2502.02379","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/no-metric-to-rule-them-all-toward-principled#ran","syntology_url":"https://syntology.ai/paper/2502.02379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02379"}},"official":{"repos":["aidos-lab/rings"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-iq-benchmarking-human-like-abstraction-and-1","slug":"mm-iq-benchmarking-human-like-abstraction-and-1","title":"MM-IQ: Benchmarking Human-Like Abstraction and Reasoning in Multimodal Models","date":"2025-02-02","arxiv_id":"2502.00698","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mm-iq-benchmarking-human-like-abstraction-and-1#ran","syntology_url":"https://syntology.ai/paper/2502.00698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.00698"}},"official":{"repos":["AceCHQ/MMIQ"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-quantum-reinforcement-learning","slug":"benchmarking-quantum-reinforcement-learning","title":"Benchmarking Quantum Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.15893","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2501.15893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15893"}},"official":{"repos":["nicomeyer96/qrl-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-benchmarking-and-robust-learning-for","slug":"scalable-benchmarking-and-robust-learning-for","title":"Scalable Benchmarking and Robust Learning for Noise-Free Ego-Motion and 3D Reconstruction from Noisy Video","date":"2025-01-24","arxiv_id":"2501.14319","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-benchmarking-and-robust-learning-for#ran","syntology_url":"https://syntology.ai/paper/2501.14319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14319"}},"official":{"repos":["xiaohao-xu/slam-under-perturbation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medagentbench-dataset-for-benchmarking-llms","slug":"medagentbench-dataset-for-benchmarking-llms","title":"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents","date":"2025-01-24","arxiv_id":"2501.14654","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/medagentbench-dataset-for-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2501.14654","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.14654"}},"official":{"repos":["stanfordmlgroup/medagentbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vchitect-2-0-parallel-transformer-for-scaling","slug":"vchitect-2-0-parallel-transformer-for-scaling","title":"Vchitect-2.0: Parallel Transformer for Scaling Up Video Diffusion Models","date":"2025-01-14","arxiv_id":"2501.08453","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vchitect-2-0-parallel-transformer-for-scaling#ran","syntology_url":"https://syntology.ai/paper/2501.08453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08453"}},"official":null}},{"url":"/paper/stronger-than-you-think-benchmarking-weak","slug":"stronger-than-you-think-benchmarking-weak","title":"Stronger Than You Think: Benchmarking Weak Supervision on Realistic Tasks","date":"2025-01-13","arxiv_id":"2501.07727","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stronger-than-you-think-benchmarking-weak#ran","syntology_url":"https://syntology.ai/paper/2501.07727","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07727"}},"official":{"repos":["jeffreywpli/stronger-than-you-think"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/voxeval-benchmarking-the-knowledge","slug":"voxeval-benchmarking-the-knowledge","title":"VoxEval: Benchmarking the Knowledge Understanding Capabilities of End-to-End Spoken Language Models","date":"2025-01-09","arxiv_id":"2501.04962","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/voxeval-benchmarking-the-knowledge#ran","syntology_url":"https://syntology.ai/paper/2501.04962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.04962"}},"official":{"repos":["dreamtheater123/voxeval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ovo-bench-how-far-is-your-video-llms-from","slug":"ovo-bench-how-far-is-your-video-llms-from","title":"OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?","date":"2025-01-09","arxiv_id":"2501.05510","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ovo-bench-how-far-is-your-video-llms-from#ran","syntology_url":"https://syntology.ai/paper/2501.05510","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05510"}},"official":{"repos":["joeleelyf/ovo-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ocrbench-v2-an-improved-benchmark-for","slug":"ocrbench-v2-an-improved-benchmark-for","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","date":"2024-12-31","arxiv_id":"2501.00321","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocrbench-v2-an-improved-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2501.00321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00321"}},"official":{"repos":["yuliang-liu/multimodalocr"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dora-sampling-and-benchmarking-for-3d-shape-1","slug":"dora-sampling-and-benchmarking-for-3d-shape-1","title":"Dora: Sampling and Benchmarking for 3D Shape Variational Auto-Encoders","date":"2024-12-23","arxiv_id":"2412.17808","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dora-sampling-and-benchmarking-for-3d-shape-1#ran","syntology_url":"https://syntology.ai/paper/2412.17808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.17808"}},"official":{"repos":["Seed3D/Dora"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/autotrust-benchmarking-trustworthiness-in","slug":"autotrust-benchmarking-trustworthiness-in","title":"AutoTrust: Benchmarking Trustworthiness in Large Vision Language Models for Autonomous Driving","date":"2024-12-19","arxiv_id":"2412.15206","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autotrust-benchmarking-trustworthiness-in#ran","syntology_url":"https://syntology.ai/paper/2412.15206","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15206"}},"official":{"repos":["taco-group/autotrust"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/antileak-bench-preventing-data-contamination","slug":"antileak-bench-preventing-data-contamination","title":"AntiLeak-Bench: Preventing Data Contamination by Automatically Constructing Benchmarks with Updated Real-World Knowledge","date":"2024-12-18","arxiv_id":"2412.13670","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/antileak-bench-preventing-data-contamination#ran","syntology_url":"https://syntology.ai/paper/2412.13670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.13670"}},"official":{"repos":["bobxwu/antileak-bench"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/theagentcompany-benchmarking-llm-agents-on","slug":"theagentcompany-benchmarking-llm-agents-on","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","date":"2024-12-18","arxiv_id":"2412.14161","repositories_listed":2,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/theagentcompany-benchmarking-llm-agents-on#ran","syntology_url":"https://syntology.ai/paper/2412.14161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.14161"}},"official":{"repos":["theagentcompany/experiments","theagentcompany/theagentcompany"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-and-understanding-compositional","slug":"benchmarking-and-understanding-compositional","title":"Benchmarking and Understanding Compositional Relational Reasoning of LLMs","date":"2024-12-17","arxiv_id":"2412.12841","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/benchmarking-and-understanding-compositional#ran","syntology_url":"https://syntology.ai/paper/2412.12841","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.12841"}},"official":{"repos":["caiyun-ai/gar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/ad-llm-benchmarking-large-language-models-for","slug":"ad-llm-benchmarking-large-language-models-for","title":"AD-LLM: Benchmarking Large Language Models for Anomaly Detection","date":"2024-12-15","arxiv_id":"2412.11142","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ad-llm-benchmarking-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2412.11142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11142"}},"official":{"repos":["usc-fortis/ad-llm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evalgim-a-library-for-evaluating-generative","slug":"evalgim-a-library-for-evaluating-generative","title":"EvalGIM: A Library for Evaluating Generative Image Models","date":"2024-12-13","arxiv_id":"2412.10604","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evalgim-a-library-for-evaluating-generative#ran","syntology_url":"https://syntology.ai/paper/2412.10604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10604"}},"official":{"repos":["facebookresearch/evalgim"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-vision-language-models-via","slug":"benchmarking-large-vision-language-models-via","title":"Benchmarking Large Vision-Language Models via Directed Scene Graph for Comprehensive Image Captioning","date":"2024-12-11","arxiv_id":"2412.08614","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2412.08614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.08614"}},"official":{"repos":["lufan31/comprecap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omnidocbench-benchmarking-diverse-pdf","slug":"omnidocbench-benchmarking-diverse-pdf","title":"OmniDocBench: Benchmarking Diverse PDF Document Parsing with Comprehensive Annotations","date":"2024-12-10","arxiv_id":"2412.07626","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/omnidocbench-benchmarking-diverse-pdf#ran","syntology_url":"https://syntology.ai/paper/2412.07626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07626"}},"official":{"repos":["opendatalab/OmniDocBench"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-browsergym-ecosystem-for-web-agent","slug":"the-browsergym-ecosystem-for-web-agent","title":"The BrowserGym Ecosystem for Web Agent Research","date":"2024-12-06","arxiv_id":"2412.05467","repositories_listed":3,"syntology":{"n":24,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":24,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/the-browsergym-ecosystem-for-web-agent#ran","syntology_url":"https://syntology.ai/paper/2412.05467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05467"}},"official":{"repos":["servicenow/agentlab","servicenow/browsergym"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/prithvi-eo-2-0-a-versatile-multi-temporal","slug":"prithvi-eo-2-0-a-versatile-multi-temporal","title":"Prithvi-EO-2.0: A Versatile Multi-Temporal Foundation Model for Earth Observation Applications","date":"2024-12-03","arxiv_id":"2412.02732","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prithvi-eo-2-0-a-versatile-multi-temporal#ran","syntology_url":"https://syntology.ai/paper/2412.02732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.02732"}},"official":{"repos":["NASA-IMPACT/Prithvi-EO-2.0"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/commit0-library-generation-from-scratch","slug":"commit0-library-generation-from-scratch","title":"Commit0: Library Generation from Scratch","date":"2024-12-02","arxiv_id":"2412.01769","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/commit0-library-generation-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2412.01769","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01769"}},"official":{"repos":["commit-0/commit0"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/geobench-vlm-benchmarking-vision-language","slug":"geobench-vlm-benchmarking-vision-language","title":"GEOBench-VLM: Benchmarking Vision-Language Models for Geospatial Tasks","date":"2024-11-28","arxiv_id":"2411.19325","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/geobench-vlm-benchmarking-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.19325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19325"}},"official":{"repos":["the-ai-alliance/geo-bench-vlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/coreval-a-comprehensive-and-objective","slug":"coreval-a-comprehensive-and-objective","title":"CHOICE: Benchmarking the Remote Sensing Capabilities of Large Vision-Language Models","date":"2024-11-27","arxiv_id":"2411.18145","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/coreval-a-comprehensive-and-objective#ran","syntology_url":"https://syntology.ai/paper/2411.18145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18145"}},"official":{"repos":["shawnan-whu/choice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-gpt-4-against-human-translators","slug":"benchmarking-gpt-4-against-human-translators","title":"Benchmarking GPT-4 against Human Translators: A Comprehensive Evaluation Across Languages, Domains, and Expertise Levels","date":"2024-11-21","arxiv_id":"2411.13775","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-gpt-4-against-human-translators#ran","syntology_url":"https://syntology.ai/paper/2411.13775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13775"}},"official":{"repos":["elliottyan/gpt_versus_mt_experts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"0ef22f27ca4255becd35eca481daceb792f1616f1ee4fc4fbee50a8c6db36eee","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}