{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/benchmarking/papers/ran/5","list_of":"/task/benchmarking","task":"Benchmarking","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":5,"pages_in_order":8,"rows_per_page":100,"rows":[401,500],"of":749,"counts":{"archive_papers_tagged":5548,"with_a_code_link":2658,"where_syntology_ran_a_sample":749,"not_listed_spam_title":0,"listed":5548,"listed_where_code_ran":749,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":624,"every_run_a_failure_of_syntologys_instrument":125,"listed_with_a_run_with_no_instrument_failure":624,"listed_every_run_a_failure_of_syntologys_instrument":125,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/benchmarking/papers/ran/1","prev":"/task/benchmarking/papers/ran/4","next":"/task/benchmarking/papers/ran/6","papers":[{"url":"/paper/do-localization-methods-actually-localize","slug":"do-localization-methods-actually-localize","title":"Do Localization Methods Actually Localize Memorized Data in LLMs? A Tale of Two Benchmarks","date":"2023-11-15","arxiv_id":"2311.09060","repositories_listed":2,"syntology":{"n":15,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/do-localization-methods-actually-localize#ran","syntology_url":"https://syntology.ai/paper/2311.09060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09060"}},"official":{"repos":["terarachang/memdata","terarachang/mempi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/magic-benchmarking-large-language-model","slug":"magic-benchmarking-large-language-model","title":"MAgIC: Investigation of Large Language Model Powered Multi-Agent in Cognition, Adaptability, Rationality and Collaboration","date":"2023-11-14","arxiv_id":"2311.08562","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/magic-benchmarking-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2311.08562","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08562"}},"official":{"repos":["cathyxl/magic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/waterbench-towards-holistic-evaluation-of","slug":"waterbench-towards-holistic-evaluation-of","title":"WaterBench: Towards Holistic Evaluation of Watermarks for Large Language Models","date":"2023-11-13","arxiv_id":"2311.07138","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":2,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/waterbench-towards-holistic-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2311.07138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07138"}},"official":{"repos":["THU-KEG/WaterBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/rethinking-and-benchmarking-predict-then","slug":"rethinking-and-benchmarking-predict-then","title":"Benchmarking PtO and PnO Methods in the Predictive Combinatorial Optimization Regime","date":"2023-11-13","arxiv_id":"2311.07633","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-and-benchmarking-predict-then#ran","syntology_url":"https://syntology.ai/paper/2311.07633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07633"}},"official":{"repos":["thinklab-sjtu/predictiveco-benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/combinatorial-optimization-with-policy","slug":"combinatorial-optimization-with-policy","title":"Combinatorial Optimization with Policy Adaptation using Latent Space Search","date":"2023-11-13","arxiv_id":"2311.13569","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/combinatorial-optimization-with-policy#ran","syntology_url":"https://syntology.ai/paper/2311.13569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13569"}},"official":{"repos":["instadeepai/compass"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-petshop-dataset-finding-causes-of","slug":"the-petshop-dataset-finding-causes-of","title":"The PetShop Dataset -- Finding Causes of Performance Issues across Microservices","date":"2023-11-08","arxiv_id":"2311.04806","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-petshop-dataset-finding-causes-of#ran","syntology_url":"https://syntology.ai/paper/2311.04806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04806"}},"official":{"repos":["amazon-science/petshop-root-cause-analysis"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fragxsitedti-revealing-responsible-segments","slug":"fragxsitedti-revealing-responsible-segments","title":"FragXsiteDTI: Revealing Responsible Segments in Drug-Target Interaction with Transformer-Driven Interpretation","date":"2023-11-04","arxiv_id":"2311.02326","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fragxsitedti-revealing-responsible-segments#ran","syntology_url":"https://syntology.ai/paper/2311.02326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02326"}},"official":{"repos":["yazdanimehdi/fragxsitedti"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/neuroevobench-benchmarking-evolutionary","slug":"neuroevobench-benchmarking-evolutionary","title":"NeuroEvoBench: Benchmarking Evolutionary Optimizers for Deep Learning Applications","date":"2023-11-04","arxiv_id":"2311.02394","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neuroevobench-benchmarking-evolutionary#ran","syntology_url":"https://syntology.ai/paper/2311.02394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02394"}},"official":{"repos":["neuroevobench/neuroevobench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/locomujoco-a-comprehensive-imitation-learning","slug":"locomujoco-a-comprehensive-imitation-learning","title":"LocoMuJoCo: A Comprehensive Imitation Learning Benchmark for Locomotion","date":"2023-11-04","arxiv_id":"2311.02496","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/locomujoco-a-comprehensive-imitation-learning#ran","syntology_url":"https://syntology.ai/paper/2311.02496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.02496"}},"official":{"repos":["robfiras/loco-mujoco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/in-search-of-lost-online-test-time-adaptation","slug":"in-search-of-lost-online-test-time-adaptation","title":"In Search of Lost Online Test-time Adaptation: A Survey","date":"2023-10-31","arxiv_id":"2310.20199","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":7,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/in-search-of-lost-online-test-time-adaptation#ran","syntology_url":"https://syntology.ai/paper/2310.20199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20199"}},"official":{"repos":["jo-wang/otta_vit_survey"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/what-s-in-my-big-data","slug":"what-s-in-my-big-data","title":"What's In My Big Data?","date":"2023-10-31","arxiv_id":"2310.20707","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-s-in-my-big-data#ran","syntology_url":"https://syntology.ai/paper/2310.20707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20707"}},"official":{"repos":["allenai/wimbd"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/battle-of-the-backbones-a-large-scale","slug":"battle-of-the-backbones-a-large-scale","title":"Battle of the Backbones: A Large-Scale Comparison of Pretrained Models across Computer Vision Tasks","date":"2023-10-30","arxiv_id":"2310.19909","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/battle-of-the-backbones-a-large-scale#ran","syntology_url":"https://syntology.ai/paper/2310.19909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19909"}},"official":{"repos":["hsouri/battle-of-the-backbones"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/bless-benchmarking-large-language-models-on","slug":"bless-benchmarking-large-language-models-on","title":"BLESS: Benchmarking Large Language Models on Sentence Simplification","date":"2023-10-24","arxiv_id":"2310.15773","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bless-benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2310.15773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15773"}},"official":{"repos":["zurichnlp/bless"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mlfmf-data-sets-for-machine-learning-for-1","slug":"mlfmf-data-sets-for-machine-learning-for-1","title":"MLFMF: Data Sets for Machine Learning for Mathematical Formalization","date":"2023-10-24","arxiv_id":"2310.16005","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mlfmf-data-sets-for-machine-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2310.16005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16005"}},"official":{"repos":["ul-fmf/mlfmf-data"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/crow-benchmarking-commonsense-reasoning-in","slug":"crow-benchmarking-commonsense-reasoning-in","title":"CRoW: Benchmarking Commonsense Reasoning in Real-World Tasks","date":"2023-10-23","arxiv_id":"2310.15239","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crow-benchmarking-commonsense-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2310.15239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15239"}},"official":{"repos":["mismayil/crow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/medeval-a-multi-level-multi-task-and-multi","slug":"medeval-a-multi-level-multi-task-and-multi","title":"MedEval: A Multi-Level, Multi-Task, and Multi-Domain Medical Benchmark for Language Model Evaluation","date":"2023-10-21","arxiv_id":"2310.14088","repositories_listed":0,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/medeval-a-multi-level-multi-task-and-multi#ran","syntology_url":"https://syntology.ai/paper/2310.14088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14088"}},"official":null}},{"url":"/paper/benchmarking-and-improving-text-to-sql","slug":"benchmarking-and-improving-text-to-sql","title":"Benchmarking and Improving Text-to-SQL Generation under Ambiguity","date":"2023-10-20","arxiv_id":"2310.13659","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-and-improving-text-to-sql#ran","syntology_url":"https://syntology.ai/paper/2310.13659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13659"}},"official":{"repos":["testzer0/ambiqt"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/oodrobustbench-benchmarking-and-analyzing","slug":"oodrobustbench-benchmarking-and-analyzing","title":"OODRobustBench: a Benchmark and Large-Scale Analysis of Adversarial Robustness under Distribution Shift","date":"2023-10-19","arxiv_id":"2310.12793","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/oodrobustbench-benchmarking-and-analyzing#ran","syntology_url":"https://syntology.ai/paper/2310.12793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12793"}},"official":{"repos":["oodrobustbench/oodrobustbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/to-generate-or-not-safety-driven-unlearned","slug":"to-generate-or-not-safety-driven-unlearned","title":"To Generate or Not? Safety-Driven Unlearned Diffusion Models Are Still Easy To Generate Unsafe Images ... For Now","date":"2023-10-18","arxiv_id":"2310.11868","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/to-generate-or-not-safety-driven-unlearned#ran","syntology_url":"https://syntology.ai/paper/2310.11868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11868"}},"official":{"repos":["optml-group/diffusion-mu-attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unveiling-the-siren-s-song-towards-reliable","slug":"unveiling-the-siren-s-song-towards-reliable","title":"FactCHD: Benchmarking Fact-Conflicting Hallucination Detection","date":"2023-10-18","arxiv_id":"2310.12086","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unveiling-the-siren-s-song-towards-reliable#ran","syntology_url":"https://syntology.ai/paper/2310.12086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12086"}},"official":{"repos":["zjunlp/factchd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/object-aware-inversion-and-reassembly-for","slug":"object-aware-inversion-and-reassembly-for","title":"Object-aware Inversion and Reassembly for Image Editing","date":"2023-10-18","arxiv_id":"2310.12149","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/object-aware-inversion-and-reassembly-for#ran","syntology_url":"https://syntology.ai/paper/2310.12149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12149"}},"official":{"repos":["aim-uofa/OIR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dialoguellm-context-and-emotion-knowledge","slug":"dialoguellm-context-and-emotion-knowledge","title":"DialogueLLM: Context and Emotion Knowledge-Tuned Large Language Models for Emotion Recognition in Conversations","date":"2023-10-17","arxiv_id":"2310.11374","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":2,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dialoguellm-context-and-emotion-knowledge#ran","syntology_url":"https://syntology.ai/paper/2310.11374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11374"}},"official":{"repos":["Dreamyao516/DialogueLLM"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evalcrafter-benchmarking-and-evaluating-large","slug":"evalcrafter-benchmarking-and-evaluating-large","title":"EvalCrafter: Benchmarking and Evaluating Large Video Generation Models","date":"2023-10-17","arxiv_id":"2310.11440","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/evalcrafter-benchmarking-and-evaluating-large#ran","syntology_url":"https://syntology.ai/paper/2310.11440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11440"}},"official":{"repos":["EvalCrafter/EvalCrafter"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/trigo-benchmarking-formal-mathematical-proof","slug":"trigo-benchmarking-formal-mathematical-proof","title":"TRIGO: Benchmarking Formal Mathematical Proof Reduction for Generative Language Models","date":"2023-10-16","arxiv_id":"2310.10180","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/trigo-benchmarking-formal-mathematical-proof#ran","syntology_url":"https://syntology.ai/paper/2310.10180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10180"}},"official":{"repos":["menik1126/TRIGO"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mirage-model-agnostic-graph-distillation-for","slug":"mirage-model-agnostic-graph-distillation-for","title":"Mirage: Model-Agnostic Graph Distillation for Graph Classification","date":"2023-10-14","arxiv_id":"2310.09486","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mirage-model-agnostic-graph-distillation-for#ran","syntology_url":"https://syntology.ai/paper/2310.09486","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09486"}},"official":{"repos":["idea-iitd/mirage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/welfare-diplomacy-benchmarking-language-model","slug":"welfare-diplomacy-benchmarking-language-model","title":"Welfare Diplomacy: Benchmarking Language Model Cooperation","date":"2023-10-13","arxiv_id":"2310.08901","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/welfare-diplomacy-benchmarking-language-model#ran","syntology_url":"https://syntology.ai/paper/2310.08901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08901"}},"official":{"repos":["mukobi/welfare-diplomacy"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kelly-is-a-warm-person-joseph-is-a-role-model","slug":"kelly-is-a-warm-person-joseph-is-a-role-model","title":"\"Kelly is a Warm Person, Joseph is a Role Model\": Gender Biases in LLM-Generated Reference Letters","date":"2023-10-13","arxiv_id":"2310.09219","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kelly-is-a-warm-person-joseph-is-a-role-model#ran","syntology_url":"https://syntology.ai/paper/2310.09219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09219"}},"official":{"repos":["uclanlp/biases-llm-reference-letters"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/metabox-a-benchmark-platform-for-meta-black-1","slug":"metabox-a-benchmark-platform-for-meta-black-1","title":"MetaBox: A Benchmark Platform for Meta-Black-Box Optimization with Reinforcement Learning","date":"2023-10-12","arxiv_id":"2310.08252","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/metabox-a-benchmark-platform-for-meta-black-1#ran","syntology_url":"https://syntology.ai/paper/2310.08252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08252"}},"official":{"repos":["GMC-DRL/MetaBox"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mcu-a-task-centric-framework-for-open-ended","slug":"mcu-a-task-centric-framework-for-open-ended","title":"Towards Evaluating Generalist Agents: An Automated Benchmark in Open World","date":"2023-10-12","arxiv_id":"2310.08367","repositories_listed":1,"syntology":{"n":19,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":19,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/mcu-a-task-centric-framework-for-open-ended#ran","syntology_url":"https://syntology.ai/paper/2310.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08367"}},"official":{"repos":["craftjarvis/mcu"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/octopus-embodied-vision-language-programmer#ran","syntology_url":"https://syntology.ai/paper/2310.08588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08588"}},"official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-progress-in-multivariate-time","slug":"exploring-progress-in-multivariate-time","title":"Exploring Progress in Multivariate Time Series Forecasting: Comprehensive Benchmarking and Heterogeneity Analysis","date":"2023-10-09","arxiv_id":"2310.06119","repositories_listed":5,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-progress-in-multivariate-time#ran","syntology_url":"https://syntology.ai/paper/2310.06119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06119"}},"official":{"repos":["gestaltcogteam/basicts","zezhishao/basicts"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/hi-guys-or-hi-folks-benchmarking-gender","slug":"hi-guys-or-hi-folks-benchmarking-gender","title":"Hi Guys or Hi Folks? Benchmarking Gender-Neutral Machine Translation with the GeNTE Corpus","date":"2023-10-08","arxiv_id":"2310.05294","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hi-guys-or-hi-folks-benchmarking-gender#ran","syntology_url":"https://syntology.ai/paper/2310.05294","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05294"}},"official":{"repos":["hlt-mt/fbk-neutr-eval"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fingpt-instruction-tuning-benchmark-for-open","slug":"fingpt-instruction-tuning-benchmark-for-open","title":"FinGPT: Instruction Tuning Benchmark for Open-Source Large Language Models in Financial Datasets","date":"2023-10-07","arxiv_id":"2310.04793","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fingpt-instruction-tuning-benchmark-for-open#ran","syntology_url":"https://syntology.ai/paper/2310.04793","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04793"}},"official":{"repos":["ai4finance-foundation/fingpt"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cifar-10-warehouse-broad-and-more-realistic","slug":"cifar-10-warehouse-broad-and-more-realistic","title":"CIFAR-10-Warehouse: Broad and More Realistic Testbeds in Model Generalization Analysis","date":"2023-10-06","arxiv_id":"2310.04414","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cifar-10-warehouse-broad-and-more-realistic#ran","syntology_url":"https://syntology.ai/paper/2310.04414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04414"}},"official":null}},{"url":"/paper/benchmarking-large-language-models-as-ai","slug":"benchmarking-large-language-models-as-ai","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","date":"2023-10-05","arxiv_id":"2310.03302","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-large-language-models-as-ai#ran","syntology_url":"https://syntology.ai/paper/2310.03302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03302"}},"official":{"repos":["snap-stanford/mlagentbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/t-3-bench-benchmarking-current-progress-in","slug":"t-3-bench-benchmarking-current-progress-in","title":"T$^3$Bench: Benchmarking Current Progress in Text-to-3D Generation","date":"2023-10-04","arxiv_id":"2310.02977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/t-3-bench-benchmarking-current-progress-in#ran","syntology_url":"https://syntology.ai/paper/2310.02977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02977"}},"official":{"repos":["THU-LYJ-Lab/T3Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causaltime-realistically-generated-time","slug":"causaltime-realistically-generated-time","title":"CausalTime: Realistically Generated Time-series for Benchmarking of Causal Discovery","date":"2023-10-03","arxiv_id":"2310.01753","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/causaltime-realistically-generated-time#ran","syntology_url":"https://syntology.ai/paper/2310.01753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01753"}},"official":{"repos":["jarrycyx/unn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/newsreclib-a-pytorch-lightning-library-for","slug":"newsreclib-a-pytorch-lightning-library-for","title":"NewsRecLib: A PyTorch-Lightning Library for Neural News Recommendation","date":"2023-10-02","arxiv_id":"2310.01146","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/newsreclib-a-pytorch-lightning-library-for#ran","syntology_url":"https://syntology.ai/paper/2310.01146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01146"}},"official":{"repos":["andreeaiana/newsreclib"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/who-is-chatgpt-benchmarking-llms","slug":"who-is-chatgpt-benchmarking-llms","title":"Who is ChatGPT? Benchmarking LLMs' Psychological Portrayal Using PsychoBench","date":"2023-10-02","arxiv_id":"2310.01386","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/who-is-chatgpt-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2310.01386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01386"}},"official":{"repos":["cuhk-arise/psychobench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-visual-scene-understanding","slug":"adaptive-visual-scene-understanding","title":"Adaptive Visual Scene Understanding: Incremental Scene Graph Generation","date":"2023-10-02","arxiv_id":"2310.01636","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adaptive-visual-scene-understanding#ran","syntology_url":"https://syntology.ai/paper/2310.01636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01636"}},"official":{"repos":["zhanglab-deepneurocoglab/csegg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-cognitive-biases-in-large","slug":"benchmarking-cognitive-biases-in-large","title":"Benchmarking Cognitive Biases in Large Language Models as Evaluators","date":"2023-09-29","arxiv_id":"2309.17012","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-cognitive-biases-in-large#ran","syntology_url":"https://syntology.ai/paper/2309.17012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17012"}},"official":{"repos":["minnesotanlp/cobbler"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fedaiot-a-federated-learning-benchmark-for","slug":"fedaiot-a-federated-learning-benchmark-for","title":"FedAIoT: A Federated Learning Benchmark for Artificial Intelligence of Things","date":"2023-09-29","arxiv_id":"2310.00109","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fedaiot-a-federated-learning-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2310.00109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00109"}},"official":{"repos":["aiot-mlsys-lab/fedaiot"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/muse-gnn-learning-unified-gene-representation","slug":"muse-gnn-learning-unified-gene-representation","title":"MuSe-GNN: Learning Unified Gene Representation From Multimodal Biological Graph Data","date":"2023-09-29","arxiv_id":"2310.02275","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/muse-gnn-learning-unified-gene-representation#ran","syntology_url":"https://syntology.ai/paper/2310.02275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02275"}},"official":{"repos":["helloworldlty/muse-gnn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-trickle-down-impact-of-reward-in","slug":"the-trickle-down-impact-of-reward-in","title":"The Trickle-down Impact of Reward (In-)consistency on RLHF","date":"2023-09-28","arxiv_id":"2309.16155","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-trickle-down-impact-of-reward-in#ran","syntology_url":"https://syntology.ai/paper/2309.16155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16155"}},"official":{"repos":["shadowkiller33/contrast-instruction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/forb-a-flat-object-retrieval-benchmark-for-1","slug":"forb-a-flat-object-retrieval-benchmark-for-1","title":"FORB: A Flat Object Retrieval Benchmark for Universal Image Embedding","date":"2023-09-28","arxiv_id":"2309.16249","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/forb-a-flat-object-retrieval-benchmark-for-1#ran","syntology_url":"https://syntology.ai/paper/2309.16249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16249"}},"official":{"repos":["pxiangwu/forb"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lawbench-benchmarking-legal-knowledge-of","slug":"lawbench-benchmarking-legal-knowledge-of","title":"LawBench: Benchmarking Legal Knowledge of Large Language Models","date":"2023-09-28","arxiv_id":"2309.16289","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lawbench-benchmarking-legal-knowledge-of#ran","syntology_url":"https://syntology.ai/paper/2309.16289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16289"}},"official":{"repos":["open-compass/lawbench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lagrangebench-a-lagrangian-fluid-mechanics-1","slug":"lagrangebench-a-lagrangian-fluid-mechanics-1","title":"LagrangeBench: A Lagrangian Fluid Mechanics Benchmarking Suite","date":"2023-09-28","arxiv_id":"2309.16342","repositories_listed":2,"syntology":{"n":27,"n_ran":23,"n_constructed":0,"n_ran_checked":17,"n_instrument":6,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":4,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lagrangebench-a-lagrangian-fluid-mechanics-1#ran","syntology_url":"https://syntology.ai/paper/2309.16342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16342"}},"official":{"repos":["tumaer/lagrangebench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","listed"]}}},{"url":"/paper/gpt-fathom-benchmarking-large-language-models","slug":"gpt-fathom-benchmarking-large-language-models","title":"GPT-Fathom: Benchmarking Large Language Models to Decipher the Evolutionary Path towards GPT-4 and Beyond","date":"2023-09-28","arxiv_id":"2309.16583","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/gpt-fathom-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2309.16583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16583"}},"official":{"repos":["gpt-fathom/gpt-fathom"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/node-aligned-graph-to-graph-generation-for","slug":"node-aligned-graph-to-graph-generation-for","title":"Node-Aligned Graph-to-Graph (NAG2G): Elevating Template-Free Deep Learning Approaches in Single-Step Retrosynthesis","date":"2023-09-27","arxiv_id":"2309.15798","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/node-aligned-graph-to-graph-generation-for#ran","syntology_url":"https://syntology.ai/paper/2309.15798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15798"}},"official":{"repos":["dptech-corp/nag2g"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/turbulence-in-focus-benchmarking-scaling","slug":"turbulence-in-focus-benchmarking-scaling","title":"Turbulence in Focus: Benchmarking Scaling Behavior of 3D Volumetric Super-Resolution with BLASTNet 2.0 Data","date":"2023-09-23","arxiv_id":"2309.13457","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/turbulence-in-focus-benchmarking-scaling#ran","syntology_url":"https://syntology.ai/paper/2309.13457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13457"}},"official":null}},{"url":"/paper/m3dsynth-a-dataset-of-medical-3d-images-with","slug":"m3dsynth-a-dataset-of-medical-3d-images-with","title":"M3Dsynth: A dataset of medical 3D images with AI-generated local manipulations","date":"2023-09-14","arxiv_id":"2309.07973","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m3dsynth-a-dataset-of-medical-3d-images-with#ran","syntology_url":"https://syntology.ai/paper/2309.07973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07973"}},"official":null}},{"url":"/paper/anchor-points-benchmarking-models-with-much","slug":"anchor-points-benchmarking-models-with-much","title":"Anchor Points: Benchmarking Models with Much Fewer Examples","date":"2023-09-14","arxiv_id":"2309.08638","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/anchor-points-benchmarking-models-with-much#ran","syntology_url":"https://syntology.ai/paper/2309.08638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.08638"}},"official":{"repos":["rvivek3/anchorpoints"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-image-dataset-for-benchmarking-recommender","slug":"an-image-dataset-for-benchmarking-recommender","title":"An Image Dataset for Benchmarking Recommender Systems with Raw Pixels","date":"2023-09-13","arxiv_id":"2309.06789","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-image-dataset-for-benchmarking-recommender#ran","syntology_url":"https://syntology.ai/paper/2309.06789","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.06789"}},"official":{"repos":["westlake-repl/pixelrec"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-skeletonization-algorithm-for-gradient","slug":"a-skeletonization-algorithm-for-gradient","title":"A skeletonization algorithm for gradient-based optimization","date":"2023-09-05","arxiv_id":"2309.02527","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/a-skeletonization-algorithm-for-gradient#ran","syntology_url":"https://syntology.ai/paper/2309.02527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.02527"}},"official":{"repos":["martinmenten/skeletonization-for-gradient-based-optimization"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-in","slug":"benchmarking-large-language-models-in","title":"Benchmarking Large Language Models in Retrieval-Augmented Generation","date":"2023-09-04","arxiv_id":"2309.01431","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2309.01431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01431"}},"official":{"repos":["chen700564/RGB"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/turbulent-flow-simulation-using","slug":"turbulent-flow-simulation-using","title":"Benchmarking Autoregressive Conditional Diffusion Models for Turbulent Flow Simulation","date":"2023-09-04","arxiv_id":"2309.01745","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/turbulent-flow-simulation-using#ran","syntology_url":"https://syntology.ai/paper/2309.01745","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01745"}},"official":{"repos":["tum-pbs/autoreg-pde-diffusion"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-quantitative-precision-for-ecg","slug":"towards-quantitative-precision-for-ecg","title":"Towards quantitative precision for ECG analysis: Leveraging state space models, self-supervision and patient metadata","date":"2023-08-29","arxiv_id":"2308.15291","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-quantitative-precision-for-ecg#ran","syntology_url":"https://syntology.ai/paper/2308.15291","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.15291"}},"official":{"repos":["tmehari/ssm_ecg"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/matbench-discovery-an-evaluation-framework","slug":"matbench-discovery-an-evaluation-framework","title":"Matbench Discovery -- A framework to evaluate machine learning crystal stability predictions","date":"2023-08-28","arxiv_id":"2308.14920","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/matbench-discovery-an-evaluation-framework#ran","syntology_url":"https://syntology.ai/paper/2308.14920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.14920"}},"official":{"repos":["janosh/matbench-discovery","janosh/pymatviz"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mllm-dataengine-an-iterative-refinement","slug":"mllm-dataengine-an-iterative-refinement","title":"MLLM-DataEngine: An Iterative Refinement Approach for MLLM","date":"2023-08-25","arxiv_id":"2308.13566","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mllm-dataengine-an-iterative-refinement#ran","syntology_url":"https://syntology.ai/paper/2308.13566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13566"}},"official":{"repos":["opendatalab/mllm-dataengine"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llmrec-benchmarking-large-language-models-on","slug":"llmrec-benchmarking-large-language-models-on","title":"LLMRec: Benchmarking Large Language Models on Recommendation Task","date":"2023-08-23","arxiv_id":"2308.12241","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":5,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llmrec-benchmarking-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2308.12241","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12241"}},"official":{"repos":["williamliujl/llmrec"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/vi-net-boosting-category-level-6d-object-pose","slug":"vi-net-boosting-category-level-6d-object-pose","title":"VI-Net: Boosting Category-level 6D Object Pose Estimation via Learning Decoupled Rotations on the Spherical Representations","date":"2023-08-19","arxiv_id":"2308.09916","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vi-net-boosting-category-level-6d-object-pose#ran","syntology_url":"https://syntology.ai/paper/2308.09916","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09916"}},"official":{"repos":["jiehonglin/vi-net"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bolaa-benchmarking-and-orchestrating-llm","slug":"bolaa-benchmarking-and-orchestrating-llm","title":"BOLAA: Benchmarking and Orchestrating LLM-augmented Autonomous Agents","date":"2023-08-11","arxiv_id":"2308.05960","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bolaa-benchmarking-and-orchestrating-llm#ran","syntology_url":"https://syntology.ai/paper/2308.05960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.05960"}},"official":{"repos":["salesforce/bolaa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/precise-benchmarking-of-explainable-ai","slug":"precise-benchmarking-of-explainable-ai","title":"Precise Benchmarking of Explainable AI Attribution Methods","date":"2023-08-06","arxiv_id":"2308.03161","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/precise-benchmarking-of-explainable-ai#ran","syntology_url":"https://syntology.ai/paper/2308.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03161"}},"official":{"repos":["rbrandt1/precise-benchmarking-of-xai"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-uncertainly-missing-and-ambiguous","slug":"rethinking-uncertainly-missing-and-ambiguous","title":"Rethinking Uncertainly Missing and Ambiguous Visual Modality in Multi-Modal Entity Alignment","date":"2023-07-30","arxiv_id":"2307.16210","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rethinking-uncertainly-missing-and-ambiguous#ran","syntology_url":"https://syntology.ai/paper/2307.16210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16210"}},"official":null}},{"url":"/paper/iml-vit-image-manipulation-localization-by","slug":"iml-vit-image-manipulation-localization-by","title":"IML-ViT: Benchmarking Image Manipulation Localization by Vision Transformer","date":"2023-07-27","arxiv_id":"2307.14863","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iml-vit-image-manipulation-localization-by#ran","syntology_url":"https://syntology.ai/paper/2307.14863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14863"}},"official":{"repos":["sunnyhaze/iml-vit"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plantain-diffusion-inspired-pose-score","slug":"plantain-diffusion-inspired-pose-score","title":"PLANTAIN: Diffusion-inspired Pose Score Minimization for Fast and Accurate Molecular Docking","date":"2023-07-22","arxiv_id":"2307.12090","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plantain-diffusion-inspired-pose-score#ran","syntology_url":"https://syntology.ai/paper/2307.12090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12090"}},"official":{"repos":["molecularmodelinglab/plantain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scibench-evaluating-college-level-scientific","slug":"scibench-evaluating-college-level-scientific","title":"SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models","date":"2023-07-20","arxiv_id":"2307.10635","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scibench-evaluating-college-level-scientific#ran","syntology_url":"https://syntology.ai/paper/2307.10635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10635"}},"official":{"repos":["mandyyyyii/scibench"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decoding-the-enigma-benchmarking-humans-and","slug":"decoding-the-enigma-benchmarking-humans-and","title":"Decoding the Enigma: Benchmarking Humans and AIs on the Many Facets of Working Memory","date":"2023-07-20","arxiv_id":"2307.10768","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoding-the-enigma-benchmarking-humans-and#ran","syntology_url":"https://syntology.ai/paper/2307.10768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10768"}},"official":{"repos":["zhanglab-deepneurocoglab/worm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/herolt-benchmarking-heterogeneous-long-tailed","slug":"herolt-benchmarking-heterogeneous-long-tailed","title":"Towards Heterogeneous Long-tailed Learning: Benchmarking, Metrics, and Toolbox","date":"2023-07-17","arxiv_id":"2307.08235","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/herolt-benchmarking-heterogeneous-long-tailed#ran","syntology_url":"https://syntology.ai/paper/2307.08235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08235"}},"official":{"repos":["ssskj/herolt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-and-accurate-optimal-transport-with","slug":"efficient-and-accurate-optimal-transport-with","title":"Efficient and Accurate Optimal Transport with Mirror Descent and Conjugate Gradients","date":"2023-07-17","arxiv_id":"2307.08507","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-and-accurate-optimal-transport-with#ran","syntology_url":"https://syntology.ai/paper/2307.08507","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08507"}},"official":{"repos":["adaptive-agents-lab/mdot-pncg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/easytpp-towards-open-benchmarking-the","slug":"easytpp-towards-open-benchmarking-the","title":"EasyTPP: Towards Open Benchmarking Temporal Point Processes","date":"2023-07-16","arxiv_id":"2307.08097","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/easytpp-towards-open-benchmarking-the#ran","syntology_url":"https://syntology.ai/paper/2307.08097","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08097"}},"official":{"repos":["ant-research/easytemporalpointprocess"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gastrovision-a-multi-class-endoscopy-image","slug":"gastrovision-a-multi-class-endoscopy-image","title":"GastroVision: A Multi-class Endoscopy Image Dataset for Computer Aided Gastrointestinal Disease Detection","date":"2023-07-16","arxiv_id":"2307.08140","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gastrovision-a-multi-class-endoscopy-image#ran","syntology_url":"https://syntology.ai/paper/2307.08140","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.08140"}},"official":{"repos":["debeshjha/gastrovision"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robotic-manipulation-datasets-for-offline","slug":"robotic-manipulation-datasets-for-offline","title":"Robotic Manipulation Datasets for Offline Compositional Reinforcement Learning","date":"2023-07-13","arxiv_id":"2307.07091","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robotic-manipulation-datasets-for-offline#ran","syntology_url":"https://syntology.ai/paper/2307.07091","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07091"}},"official":{"repos":["lifelong-ml/offline-compositional-rl-datasets"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-algorithms-for-federated-domain","slug":"benchmarking-algorithms-for-federated-domain","title":"Benchmarking Algorithms for Federated Domain Generalization","date":"2023-07-11","arxiv_id":"2307.04942","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-algorithms-for-federated-domain#ran","syntology_url":"https://syntology.ai/paper/2307.04942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.04942"}},"official":{"repos":["inouye-lab/feddg_benchmark"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-test-time-adaptation-against","slug":"benchmarking-test-time-adaptation-against","title":"Benchmarking Test-Time Adaptation against Distribution Shifts in Image Classification","date":"2023-07-06","arxiv_id":"2307.03133","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-test-time-adaptation-against#ran","syntology_url":"https://syntology.ai/paper/2307.03133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03133"}},"official":{"repos":["yuyongcan/benchmark-tta"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/intercode-standardizing-and-benchmarking","slug":"intercode-standardizing-and-benchmarking","title":"InterCode: Standardizing and Benchmarking Interactive Coding with Execution Feedback","date":"2023-06-26","arxiv_id":"2306.14898","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/intercode-standardizing-and-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2306.14898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14898"}},"official":{"repos":["princeton-nlp/intercode"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optiforest-optimal-isolation-forest-for","slug":"optiforest-optimal-isolation-forest-for","title":"OptIForest: Optimal Isolation Forest for Anomaly Detection","date":"2023-06-22","arxiv_id":"2306.12703","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":6,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 6 samples that ran constructed an object rather than computing a result","sample_list":"/paper/optiforest-optimal-isolation-forest-for#ran","syntology_url":"https://syntology.ai/paper/2306.12703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.12703"}},"official":{"repos":["xiagll/optiforest"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-express-their-uncertainty-an","slug":"can-llms-express-their-uncertainty-an","title":"Can LLMs Express Their Uncertainty? An Empirical Evaluation of Confidence Elicitation in LLMs","date":"2023-06-22","arxiv_id":"2306.13063","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/can-llms-express-their-uncertainty-an#ran","syntology_url":"https://syntology.ai/paper/2306.13063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13063"}},"official":{"repos":["miaoxiong2320/llm-uncertainty"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/did-the-models-understand-documents","slug":"did-the-models-understand-documents","title":"Did the Models Understand Documents? Benchmarking Models for Language Understanding in Document-Level Relation Extraction","date":"2023-06-20","arxiv_id":"2306.11386","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/did-the-models-understand-documents#ran","syntology_url":"https://syntology.ai/paper/2306.11386","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11386"}},"official":{"repos":["hytn/docred-hwe"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-normal-on-the-evaluation-of-mutual-1","slug":"beyond-normal-on-the-evaluation-of-mutual-1","title":"Beyond Normal: On the Evaluation of Mutual Information Estimators","date":"2023-06-19","arxiv_id":"2306.11078","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-normal-on-the-evaluation-of-mutual-1#ran","syntology_url":"https://syntology.ai/paper/2306.11078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11078"}},"official":{"repos":["cbg-ethz/bmi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openp5-benchmarking-foundation-models-for","slug":"openp5-benchmarking-foundation-models-for","title":"OpenP5: An Open-Source Platform for Developing, Training, and Evaluating LLM-based Recommender Systems","date":"2023-06-19","arxiv_id":"2306.11134","repositories_listed":4,"syntology":{"n":25,"n_ran":23,"n_constructed":0,"n_ran_checked":20,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":18,"n_pointer_only":9,"phrase":"23 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 1 honoured, 1 violated, 18 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/openp5-benchmarking-foundation-models-for#ran","syntology_url":"https://syntology.ai/paper/2306.11134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.11134"}},"official":{"repos":["agiresearch/openp5"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/evaluating-graph-neural-networks-for-link","slug":"evaluating-graph-neural-networks-for-link","title":"Evaluating Graph Neural Networks for Link Prediction: Current Pitfalls and New Benchmarking","date":"2023-06-18","arxiv_id":"2306.10453","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-graph-neural-networks-for-link#ran","syntology_url":"https://syntology.ai/paper/2306.10453","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10453"}},"official":{"repos":["juanhui28/heart"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/opendataval-a-unified-benchmark-for-data-1","slug":"opendataval-a-unified-benchmark-for-data-1","title":"OpenDataVal: a Unified Benchmark for Data Valuation","date":"2023-06-18","arxiv_id":"2306.10577","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/opendataval-a-unified-benchmark-for-data-1#ran","syntology_url":"https://syntology.ai/paper/2306.10577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10577"}},"official":{"repos":["opendataval/opendataval"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/companykg-a-large-scale-heterogeneous-graph","slug":"companykg-a-large-scale-heterogeneous-graph","title":"CompanyKG: A Large-Scale Heterogeneous Graph for Company Similarity Quantification","date":"2023-06-18","arxiv_id":"2306.10649","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/companykg-a-large-scale-heterogeneous-graph#ran","syntology_url":"https://syntology.ai/paper/2306.10649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10649"}},"official":{"repos":["eqtpartners/companykg"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pinnacle-a-comprehensive-benchmark-of-physics","slug":"pinnacle-a-comprehensive-benchmark-of-physics","title":"PINNacle: A Comprehensive Benchmark of Physics-Informed Neural Networks for Solving PDEs","date":"2023-06-15","arxiv_id":"2306.08827","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/pinnacle-a-comprehensive-benchmark-of-physics#ran","syntology_url":"https://syntology.ai/paper/2306.08827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08827"}},"official":{"repos":["i207m/pinnacle"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/datasets-and-benchmarks-for-offline-safe","slug":"datasets-and-benchmarks-for-offline-safe","title":"Datasets and Benchmarks for Offline Safe Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09303","repositories_listed":3,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/datasets-and-benchmarks-for-offline-safe#ran","syntology_url":"https://syntology.ai/paper/2306.09303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09303"}},"official":{"repos":["liuzuxin/dsrl","liuzuxin/fsrl","liuzuxin/osrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/symmetry-informed-geometric-representation-1","slug":"symmetry-informed-geometric-representation-1","title":"Symmetry-Informed Geometric Representation for Molecules, Proteins, and Crystalline Materials","date":"2023-06-15","arxiv_id":"2306.09375","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/symmetry-informed-geometric-representation-1#ran","syntology_url":"https://syntology.ai/paper/2306.09375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09375"}},"official":{"repos":["chao1224/geom3d"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ffb-a-fair-fairness-benchmark-for-in","slug":"ffb-a-fair-fairness-benchmark-for-in","title":"FFB: A Fair Fairness Benchmark for In-Processing Group Fairness Methods","date":"2023-06-15","arxiv_id":"2306.09468","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ffb-a-fair-fairness-benchmark-for-in#ran","syntology_url":"https://syntology.ai/paper/2306.09468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09468"}},"official":{"repos":["ahxt/fair_fairness_benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/muben-benchmarking-the-uncertainty-of-pre","slug":"muben-benchmarking-the-uncertainty-of-pre","title":"MUBen: Benchmarking the Uncertainty of Molecular Representation Models","date":"2023-06-14","arxiv_id":"2306.10060","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/muben-benchmarking-the-uncertainty-of-pre#ran","syntology_url":"https://syntology.ai/paper/2306.10060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.10060"}},"official":{"repos":["Yinghao-Li/UncertaintyBenchmark","yinghao-li/muben"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-neural-network-training","slug":"benchmarking-neural-network-training","title":"Benchmarking Neural Network Training Algorithms","date":"2023-06-12","arxiv_id":"2306.07179","repositories_listed":4,"syntology":{"n":22,"n_ran":21,"n_constructed":0,"n_ran_checked":21,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":21,"n_pointer_only":0,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 0 honoured, 0 violated, 21 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-neural-network-training#ran","syntology_url":"https://syntology.ai/paper/2306.07179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07179"}},"official":{"repos":["mlcommons/algorithmic-efficiency"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/neurograph-benchmarks-for-graph-machine","slug":"neurograph-benchmarks-for-graph-machine","title":"NeuroGraph: Benchmarks for Graph Machine Learning in Brain Connectomics","date":"2023-06-09","arxiv_id":"2306.06202","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/neurograph-benchmarks-for-graph-machine#ran","syntology_url":"https://syntology.ai/paper/2306.06202","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06202"}},"official":{"repos":["anwar-said/neurograph"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/yet-another-icu-benchmark-a-flexible-multi","slug":"yet-another-icu-benchmark-a-flexible-multi","title":"Yet Another ICU Benchmark: A Flexible Multi-Center Framework for Clinical ML","date":"2023-06-08","arxiv_id":"2306.05109","repositories_listed":4,"syntology":{"n":25,"n_ran":21,"n_constructed":0,"n_ran_checked":18,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/yet-another-icu-benchmark-a-flexible-multi#ran","syntology_url":"https://syntology.ai/paper/2306.05109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05109"}},"official":{"repos":["rvandewater/recipys","rvandewater/yaib","rvandewater/yaib-cohorts","rvandewater/yaib-models"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-large-language-models-on-cmexam#ran","syntology_url":"https://syntology.ai/paper/2306.03030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03030"}},"official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/libauc-a-deep-learning-library-for-x-risk","slug":"libauc-a-deep-learning-library-for-x-risk","title":"LibAUC: A Deep Learning Library for X-Risk Optimization","date":"2023-06-05","arxiv_id":"2306.03065","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/libauc-a-deep-learning-library-for-x-risk#ran","syntology_url":"https://syntology.ai/paper/2306.03065","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03065"}},"official":{"repos":["Optimization-AI/LibAUC"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/score-based-enhanced-sampling-for-protein","slug":"score-based-enhanced-sampling-for-protein","title":"Str2Str: A Score-based Framework for Zero-shot Protein Conformation Sampling","date":"2023-06-05","arxiv_id":"2306.03117","repositories_listed":1,"syntology":{"n":24,"n_ran":22,"n_constructed":0,"n_ran_checked":20,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":20,"n_pointer_only":3,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/score-based-enhanced-sampling-for-protein#ran","syntology_url":"https://syntology.ai/paper/2306.03117","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03117"}},"official":{"repos":["lujiarui/str2str"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":20,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/libero-benchmarking-knowledge-transfer-for","slug":"libero-benchmarking-knowledge-transfer-for","title":"LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning","date":"2023-06-05","arxiv_id":"2306.03310","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/libero-benchmarking-knowledge-transfer-for#ran","syntology_url":"https://syntology.ai/paper/2306.03310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03310"}},"official":null}},{"url":"/paper/multilingual-conceptual-coverage-in-text-to","slug":"multilingual-conceptual-coverage-in-text-to","title":"Multilingual Conceptual Coverage in Text-to-Image Models","date":"2023-06-02","arxiv_id":"2306.01735","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multilingual-conceptual-coverage-in-text-to#ran","syntology_url":"https://syntology.ai/paper/2306.01735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01735"}},"official":{"repos":["michaelsaxon/cococrola"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spatially-resolved-gene-expression-prediction","slug":"spatially-resolved-gene-expression-prediction","title":"Spatially Resolved Gene Expression Prediction from H&E Histology Images via Bi-modal Contrastive Learning","date":"2023-06-02","arxiv_id":"2306.01859","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatially-resolved-gene-expression-prediction#ran","syntology_url":"https://syntology.ai/paper/2306.01859","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01859"}},"official":{"repos":["bowang-lab/bleep"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-integral-losses-in-physics","slug":"learning-from-integral-losses-in-physics","title":"Learning from Integral Losses in Physics Informed Neural Networks","date":"2023-05-27","arxiv_id":"2305.17387","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":1,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-integral-losses-in-physics#ran","syntology_url":"https://syntology.ai/paper/2305.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17387"}},"official":{"repos":["ehsansaleh/btspinn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/property-guided-generative-modelling-for","slug":"property-guided-generative-modelling-for","title":"Robust Model-Based Optimization for Challenging Fitness Landscapes","date":"2023-05-23","arxiv_id":"2305.13650","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/property-guided-generative-modelling-for#ran","syntology_url":"https://syntology.ai/paper/2305.13650","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13650"}},"official":{"repos":["sabagh1994/pgvae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"a2b10a8a7574caca33914de29a6758ea4a4b2473fdd66a8bac9418364a40a4ad","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}