{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/ran/1","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":161,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice/papers/ran/1","prev":null,"next":"/task/multiple-choice/papers/ran/2","papers":[{"url":"/paper/daily-omni-towards-audio-visual-reasoning","slug":"daily-omni-towards-audio-visual-reasoning","title":"Daily-Omni: Towards Audio-Visual Reasoning with Temporal Alignment across Modalities","date":"2025-05-23","arxiv_id":"2505.17862","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/daily-omni-towards-audio-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/2505.17862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.17862"}},"official":{"repos":["lliar-liar/daily-omni"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/videoeval-pro-robust-and-realistic-long-video","slug":"videoeval-pro-robust-and-realistic-long-video","title":"VideoEval-Pro: Robust and Realistic Long Video Understanding Evaluation","date":"2025-05-20","arxiv_id":"2505.14640","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videoeval-pro-robust-and-realistic-long-video#ran","syntology_url":"https://syntology.ai/paper/2505.14640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14640"}},"official":null}},{"url":"/paper/healthbench-evaluating-large-language-models","slug":"healthbench-evaluating-large-language-models","title":"HealthBench: Evaluating Large Language Models Towards Improved Human Health","date":"2025-05-13","arxiv_id":"2505.08775","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/healthbench-evaluating-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2505.08775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.08775"}},"official":{"repos":["openai/simple-evals"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chartqapro-a-more-diverse-and-challenging","slug":"chartqapro-a-more-diverse-and-challenging","title":"ChartQAPro: A More Diverse and Challenging Benchmark for Chart Question Answering","date":"2025-04-07","arxiv_id":"2504.05506","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartqapro-a-more-diverse-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2504.05506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05506"}},"official":{"repos":["vis-nlp/chartqapro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-the-effect-of-reinforcement","slug":"exploring-the-effect-of-reinforcement","title":"Exploring the Effect of Reinforcement Learning on Video Understanding: Insights from SEED-Bench-R1","date":"2025-03-31","arxiv_id":"2503.24376","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-the-effect-of-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.24376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.24376"}},"official":{"repos":["tencentarc/seed-bench-r1"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hybrid-level-instruction-injection-for-video","slug":"hybrid-level-instruction-injection-for-video","title":"Hybrid-Level Instruction Injection for Video Token Compression in Multi-modal Large Language Models","date":"2025-03-20","arxiv_id":"2503.16036","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hybrid-level-instruction-injection-for-video#ran","syntology_url":"https://syntology.ai/paper/2503.16036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16036"}},"official":{"repos":["lntzm/hicom"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mellow-a-small-audio-language-model-for","slug":"mellow-a-small-audio-language-model-for","title":"Mellow: a small audio language model for reasoning","date":"2025-03-11","arxiv_id":"2503.08540","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mellow-a-small-audio-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2503.08540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08540"}},"official":{"repos":["soham97/mellow"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/visbias-measuring-explicit-and-implicit","slug":"visbias-measuring-explicit-and-implicit","title":"VisBias: Measuring Explicit and Implicit Social Biases in Vision Language Models","date":"2025-03-10","arxiv_id":"2503.07575","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visbias-measuring-explicit-and-implicit#ran","syntology_url":"https://syntology.ai/paper/2503.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07575"}},"official":{"repos":["uscnlp-lime/visbias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2503-00096","slug":"2503-00096","title":"BixBench: a Comprehensive Benchmark for LLM-based Agents in Computational Biology","date":"2025-02-28","arxiv_id":"2503.00096","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2503-00096#ran","syntology_url":"https://syntology.ai/paper/2503.00096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00096"}},"official":{"repos":["Future-House/BixBench","Future-House/data-analysis-crow"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/big-math-a-large-scale-high-quality-math","slug":"big-math-a-large-scale-high-quality-math","title":"Big-Math: A Large-Scale, High-Quality Math Dataset for Reinforcement Learning in Language Models","date":"2025-02-24","arxiv_id":"2502.17387","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/big-math-a-large-scale-high-quality-math#ran","syntology_url":"https://syntology.ai/paper/2502.17387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.17387"}},"official":{"repos":["synthlabsai/big-math"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/moving-beyond-medical-exam-questions-a","slug":"moving-beyond-medical-exam-questions-a","title":"Moving Beyond Medical Exam Questions: A Clinician-Annotated Dataset of Real-World Tasks and Ambiguity in Mental Healthcare","date":"2025-02-22","arxiv_id":"2502.16051","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/moving-beyond-medical-exam-questions-a#ran","syntology_url":"https://syntology.ai/paper/2502.16051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16051"}},"official":{"repos":["maxlampe/mentat"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-confidence-gap-overconfidence","slug":"mind-the-confidence-gap-overconfidence","title":"Mind the Confidence Gap: Overconfidence, Calibration, and Distractor Effects in Large Language Models","date":"2025-02-16","arxiv_id":"2502.11028","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-confidence-gap-overconfidence#ran","syntology_url":"https://syntology.ai/paper/2502.11028","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11028"}},"official":{"repos":["prateekchhikara/llms-calibration"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-specialized-visual-encoders-for","slug":"unifying-specialized-visual-encoders-for","title":"Unifying Specialized Visual Encoders for Video Language Models","date":"2025-01-02","arxiv_id":"2501.01426","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-specialized-visual-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2501.01426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.01426"}},"official":{"repos":["princetonvisualai/merv"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mapeval-a-map-based-evaluation-of-geo-spatial","slug":"mapeval-a-map-based-evaluation-of-geo-spatial","title":"MapEval: A Map-Based Evaluation of Geo-Spatial Reasoning in Foundation Models","date":"2024-12-31","arxiv_id":"2501.00316","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mapeval-a-map-based-evaluation-of-geo-spatial#ran","syntology_url":"https://syntology.ai/paper/2501.00316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.00316"}},"official":{"repos":["MapEval/MapEval-API","MapEval/MapEval-Textual","MapEval/MapEval-Visual"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longbench-v2-towards-deeper-understanding-and","slug":"longbench-v2-towards-deeper-understanding-and","title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","date":"2024-12-19","arxiv_id":"2412.15204","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longbench-v2-towards-deeper-understanding-and#ran","syntology_url":"https://syntology.ai/paper/2412.15204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.15204"}},"official":{"repos":["thudm/longbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-and-mitigating-social-bias-for","slug":"evaluating-and-mitigating-social-bias-for","title":"Evaluating and Mitigating Social Bias for Large Language Models in Open-ended Settings","date":"2024-12-09","arxiv_id":"2412.06134","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-and-mitigating-social-bias-for#ran","syntology_url":"https://syntology.ai/paper/2412.06134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.06134"}},"official":{"repos":["zhaoliu0914/LLM-Bias-Benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/noise-injection-reveals-hidden-capabilities","slug":"noise-injection-reveals-hidden-capabilities","title":"Noise Injection Reveals Hidden Capabilities of Sandbagging Language Models","date":"2024-12-02","arxiv_id":"2412.01784","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noise-injection-reveals-hidden-capabilities#ran","syntology_url":"https://syntology.ai/paper/2412.01784","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.01784"}},"official":{"repos":["camtice/sandbagdetect"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/coreval-a-comprehensive-and-objective","slug":"coreval-a-comprehensive-and-objective","title":"CHOICE: Benchmarking the Remote Sensing Capabilities of Large Vision-Language Models","date":"2024-11-27","arxiv_id":"2411.18145","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/coreval-a-comprehensive-and-objective#ran","syntology_url":"https://syntology.ai/paper/2411.18145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.18145"}},"official":{"repos":["shawnan-whu/choice"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/hourvideo-1-hour-video-language-understanding","slug":"hourvideo-1-hour-video-language-understanding","title":"HourVideo: 1-Hour Video-Language Understanding","date":"2024-11-07","arxiv_id":"2411.04998","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/hourvideo-1-hour-video-language-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.04998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.04998"}},"official":{"repos":["keshik6/HourVideo"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ppllava-varied-video-sequence-understanding","slug":"ppllava-varied-video-sequence-understanding","title":"PPLLaVA: Varied Video Sequence Understanding With Prompt Guidance","date":"2024-11-04","arxiv_id":"2411.02327","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/ppllava-varied-video-sequence-understanding#ran","syntology_url":"https://syntology.ai/paper/2411.02327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02327"}},"official":{"repos":["farewellthree/ppllava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/timeseriesexam-a-time-series-understanding","slug":"timeseriesexam-a-time-series-understanding","title":"TimeSeriesExam: A time series understanding exam","date":"2024-10-18","arxiv_id":"2410.14752","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timeseriesexam-a-time-series-understanding#ran","syntology_url":"https://syntology.ai/paper/2410.14752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14752"}},"official":null}},{"url":"/paper/mmie-massive-multimodal-interleaved","slug":"mmie-massive-multimodal-interleaved","title":"MMIE: Massive Multimodal Interleaved Comprehension Benchmark for Large Vision-Language Models","date":"2024-10-14","arxiv_id":"2410.10139","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmie-massive-multimodal-interleaved#ran","syntology_url":"https://syntology.ai/paper/2410.10139","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10139"}},"official":{"repos":["Lillianwei-h/MMIE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/novo-norm-voting-off-hallucinations-with","slug":"novo-norm-voting-off-hallucinations-with","title":"NoVo: Norm Voting off Hallucinations with Attention Heads in Large Language Models","date":"2024-10-11","arxiv_id":"2410.08970","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/novo-norm-voting-off-hallucinations-with#ran","syntology_url":"https://syntology.ai/paper/2410.08970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.08970"}},"official":null}},{"url":"/paper/a-hitchhikers-guide-to-fine-grained-face","slug":"a-hitchhikers-guide-to-fine-grained-face","title":"A Hitchhikers Guide to Fine-Grained Face Forgery Detection Using Common Sense Reasoning","date":"2024-10-01","arxiv_id":"2410.00485","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-hitchhikers-guide-to-fine-grained-face#ran","syntology_url":"https://syntology.ai/paper/2410.00485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.00485"}},"official":{"repos":["NickyFot/HitchhikersGuide"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/columbus-evaluating-cognitive-lateral","slug":"columbus-evaluating-cognitive-lateral","title":"COLUMBUS: Evaluating COgnitive Lateral Understanding through Multiple-choice reBUSes","date":"2024-09-06","arxiv_id":"2409.04053","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/columbus-evaluating-cognitive-lateral#ran","syntology_url":"https://syntology.ai/paper/2409.04053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04053"}},"official":{"repos":["koen-47/columbus"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","slug":"cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","arxiv_id":"2409.02834","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to#ran","syntology_url":"https://syntology.ai/paper/2409.02834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.02834"}},"official":{"repos":["ecnu-icalk/educhat-math"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/training-on-the-benchmark-is-not-all-you-need","slug":"training-on-the-benchmark-is-not-all-you-need","title":"Training on the Benchmark Is Not All You Need","date":"2024-09-03","arxiv_id":"2409.01790","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/training-on-the-benchmark-is-not-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2409.01790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01790"}},"official":{"repos":["nishiwen1214/Benchmark-leakage-detection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wait-that-s-not-an-option-llms-robustness","slug":"wait-that-s-not-an-option-llms-robustness","title":"Wait, that's not an option: LLMs Robustness with Incorrect Multiple-Choice Options","date":"2024-08-27","arxiv_id":"2409.00113","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wait-that-s-not-an-option-llms-robustness#ran","syntology_url":"https://syntology.ai/paper/2409.00113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00113"}},"official":{"repos":["gracjangoral/when-all-options-are-wrong"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-evaluating-and-building-versatile","slug":"towards-evaluating-and-building-versatile","title":"Towards Evaluating and Building Versatile Large Language Models for Medicine","date":"2024-08-22","arxiv_id":"2408.12547","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-evaluating-and-building-versatile#ran","syntology_url":"https://syntology.ai/paper/2408.12547","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12547"}},"official":{"repos":["magic-ai4med/meds-ins"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-visual-sycophancy-in-multimodal","slug":"measuring-visual-sycophancy-in-multimodal","title":"Measuring Agreeableness Bias in Multimodal Models","date":"2024-08-17","arxiv_id":"2408.09111","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-visual-sycophancy-in-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.09111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09111"}},"official":{"repos":["jasonlim131/looksRdeceiving"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llms-are-biased-towards-output-formats","slug":"llms-are-biased-towards-output-formats","title":"LLMs Are Biased Towards Output Formats! Systematically Evaluating and Mitigating Output Format Bias of LLMs","date":"2024-08-16","arxiv_id":"2408.08656","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llms-are-biased-towards-output-formats#ran","syntology_url":"https://syntology.ai/paper/2408.08656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08656"}},"official":{"repos":["dxlong2000/FormatEval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01800","slug":"2408-01800","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","date":"2024-08-03","arxiv_id":"2408.01800","repositories_listed":2,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-01800#ran","syntology_url":"https://syntology.ai/paper/2408.01800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01800"}},"official":{"repos":["OpenBMB/MiniCPM-o"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/2408-01337","slug":"2408-01337","title":"MuChoMusic: Evaluating Music Understanding in Multimodal Audio-Language Models","date":"2024-08-02","arxiv_id":"2408.01337","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-01337#ran","syntology_url":"https://syntology.ai/paper/2408.01337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01337"}},"official":{"repos":["mulab-mir/muchomusic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/annealed-multiple-choice-learning-overcoming","slug":"annealed-multiple-choice-learning-overcoming","title":"Annealed Multiple Choice Learning: Overcoming limitations of Winner-takes-all with annealing","date":"2024-07-22","arxiv_id":"2407.15580","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/annealed-multiple-choice-learning-overcoming#ran","syntology_url":"https://syntology.ai/paper/2407.15580","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15580"}},"official":{"repos":["victorletzelter/annealed_mcl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longvideobench-a-benchmark-for-long-context","slug":"longvideobench-a-benchmark-for-long-context","title":"LongVideoBench: A Benchmark for Long-context Interleaved Video-Language Understanding","date":"2024-07-22","arxiv_id":"2407.15754","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longvideobench-a-benchmark-for-long-context#ran","syntology_url":"https://syntology.ai/paper/2407.15754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15754"}},"official":{"repos":["longvideobench/longvideobench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mminstruct-a-high-quality-multi-modal","slug":"mminstruct-a-high-quality-multi-modal","title":"MMInstruct: A High-Quality Multi-Modal Instruction Tuning Dataset with Extensive Diversity","date":"2024-07-22","arxiv_id":"2407.15838","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mminstruct-a-high-quality-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2407.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15838"}},"official":{"repos":["yuecao0119/mminstruct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-assemble-ace-understanding-how","slug":"answer-assemble-ace-understanding-how","title":"Answer, Assemble, Ace: Understanding How Transformers Answer Multiple Choice Questions","date":"2024-07-21","arxiv_id":"2407.15018","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-assemble-ace-understanding-how#ran","syntology_url":"https://syntology.ai/paper/2407.15018","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15018"}},"official":null}},{"url":"/paper/generalization-v-s-memorization-tracing","slug":"generalization-v-s-memorization-tracing","title":"Generalization v.s. Memorization: Tracing Language Models' Capabilities Back to Pretraining Data","date":"2024-07-20","arxiv_id":"2407.14985","repositories_listed":0,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/generalization-v-s-memorization-tracing#ran","syntology_url":"https://syntology.ai/paper/2407.14985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14985"}},"official":null}},{"url":"/paper/evaluating-language-models-as-risk-scores","slug":"evaluating-language-models-as-risk-scores","title":"Evaluating language models as risk scores","date":"2024-07-19","arxiv_id":"2407.14614","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/evaluating-language-models-as-risk-scores#ran","syntology_url":"https://syntology.ai/paper/2407.14614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14614"}},"official":{"repos":["socialfoundations/folktexts"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-large-language-models-for","slug":"harnessing-large-language-models-for","title":"Fine-tuning Multimodal Large Language Models for Product Bundling","date":"2024-07-16","arxiv_id":"2407.11712","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2407.11712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11712"}},"official":{"repos":["xiaohao-liu/bundle-mllm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-is-fragile-manipulating","slug":"uncertainty-is-fragile-manipulating","title":"Uncertainty is Fragile: Manipulating Uncertainty in Large Language Models","date":"2024-07-15","arxiv_id":"2407.11282","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/uncertainty-is-fragile-manipulating#ran","syntology_url":"https://syntology.ai/paper/2407.11282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11282"}},"official":{"repos":["qcznlp/uncertainty_attack","qcznlp/uncertainty"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mmsci-a-multimodal-multi-discipline-dataset","slug":"mmsci-a-multimodal-multi-discipline-dataset","title":"MMSci: A Dataset for Graduate-Level Multi-Discipline Multimodal Scientific Understanding","date":"2024-07-06","arxiv_id":"2407.04903","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":1,"n_violates":1,"n_no_contract":7,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mmsci-a-multimodal-multi-discipline-dataset#ran","syntology_url":"https://syntology.ai/paper/2407.04903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04903"}},"official":{"repos":["leezekun/mmsci"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/logicvista-multimodal-llm-logical-reasoning","slug":"logicvista-multimodal-llm-logical-reasoning","title":"LogicVista: Multimodal LLM Logical Reasoning Benchmark in Visual Contexts","date":"2024-07-06","arxiv_id":"2407.04973","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/logicvista-multimodal-llm-logical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.04973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04973"}},"official":{"repos":["yijia-xiao/logicvista"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-consistent-over","slug":"are-large-language-models-consistent-over","title":"Are Large Language Models Consistent over Value-laden Questions?","date":"2024-07-03","arxiv_id":"2407.02996","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-large-language-models-consistent-over#ran","syntology_url":"https://syntology.ai/paper/2407.02996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02996"}},"official":{"repos":["jlcmoore/ValueConsistency"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/divert-distractor-generation-with-variational","slug":"divert-distractor-generation-with-variational","title":"DiVERT: Distractor Generation with Variational Errors Represented as Text for Math Multiple-choice Questions","date":"2024-06-27","arxiv_id":"2406.19356","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/divert-distractor-generation-with-variational#ran","syntology_url":"https://syntology.ai/paper/2406.19356","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19356"}},"official":{"repos":["umass-ml4ed/divert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/varbench-robust-language-model-benchmarking","slug":"varbench-robust-language-model-benchmarking","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","date":"2024-06-25","arxiv_id":"2406.17681","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/varbench-robust-language-model-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.17681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17681"}},"official":{"repos":["qbetterk/VarBench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hcqa-ego4d-egoschema-challenge-2024","slug":"hcqa-ego4d-egoschema-challenge-2024","title":"HCQA @ Ego4D EgoSchema Challenge 2024","date":"2024-06-22","arxiv_id":"2406.15771","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hcqa-ego4d-egoschema-challenge-2024#ran","syntology_url":"https://syntology.ai/paper/2406.15771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15771"}},"official":{"repos":["hyu-zhang/hcqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/african-or-european-swallow-benchmarking","slug":"african-or-european-swallow-benchmarking","title":"African or European Swallow? Benchmarking Large Vision-Language Models for Fine-Grained Object Classification","date":"2024-06-20","arxiv_id":"2406.14496","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/african-or-european-swallow-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.14496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14496"}},"official":{"repos":["gregor-ge/foci-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/foodieqa-a-multimodal-dataset-for-fine","slug":"foodieqa-a-multimodal-dataset-for-fine","title":"FoodieQA: A Multimodal Dataset for Fine-Grained Understanding of Chinese Food Culture","date":"2024-06-16","arxiv_id":"2406.11030","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foodieqa-a-multimodal-dataset-for-fine#ran","syntology_url":"https://syntology.ai/paper/2406.11030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11030"}},"official":{"repos":["lyan62/FoodieQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/color-filter-conditional-loss-reduction","slug":"color-filter-conditional-loss-reduction","title":"CoLoR-Filter: Conditional Loss Reduction Filtering for Targeted Language Model Pre-training","date":"2024-06-15","arxiv_id":"2406.10670","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/color-filter-conditional-loss-reduction#ran","syntology_url":"https://syntology.ai/paper/2406.10670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10670"}},"official":{"repos":["davidbrandfonbrener/color-filter-olmo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/blend-a-benchmark-for-llms-on-everyday","slug":"blend-a-benchmark-for-llms-on-everyday","title":"BLEnD: A Benchmark for LLMs on Everyday Knowledge in Diverse Cultures and Languages","date":"2024-06-14","arxiv_id":"2406.09948","repositories_listed":1,"syntology":{"n":10,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":10,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/blend-a-benchmark-for-llms-on-everyday#ran","syntology_url":"https://syntology.ai/paper/2406.09948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09948"}},"official":{"repos":["nlee0212/blend"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/muirbench-a-comprehensive-benchmark-for","slug":"muirbench-a-comprehensive-benchmark-for","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","date":"2024-06-13","arxiv_id":"2406.09411","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/muirbench-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2406.09411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09411"}},"official":{"repos":["muirbench/MuirBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bertaqa-how-much-do-language-models-know","slug":"bertaqa-how-much-do-language-models-know","title":"BertaQA: How Much Do Language Models Know About Local Culture?","date":"2024-06-11","arxiv_id":"2406.07302","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/bertaqa-how-much-do-language-models-know#ran","syntology_url":"https://syntology.ai/paper/2406.07302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07302"}},"official":{"repos":["juletx/bertaqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/videollama-2-advancing-spatial-temporal","slug":"videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","arxiv_id":"2406.07476","repositories_listed":3,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":7,"n_instrument":5,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":6,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/videollama-2-advancing-spatial-temporal#ran","syntology_url":"https://syntology.ai/paper/2406.07476","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07476"}},"official":{"repos":["damo-nlp-sg/videollama2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/open-llm-leaderboard-from-multi-choice-to","slug":"open-llm-leaderboard-from-multi-choice-to","title":"Open-LLM-Leaderboard: From Multi-choice to Open-style Questions for LLMs Evaluation, Benchmark, and Arena","date":"2024-06-11","arxiv_id":"2406.07545","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-llm-leaderboard-from-multi-choice-to#ran","syntology_url":"https://syntology.ai/paper/2406.07545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07545"}},"official":{"repos":["vila-lab/open-llm-leaderboard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-fine-tuning-dataset-and-benchmark-for-large","slug":"a-fine-tuning-dataset-and-benchmark-for-large","title":"A Fine-tuning Dataset and Benchmark for Large Language Models for Protein Understanding","date":"2024-06-08","arxiv_id":"2406.05540","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-fine-tuning-dataset-and-benchmark-for-large#ran","syntology_url":"https://syntology.ai/paper/2406.05540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05540"}},"official":{"repos":["tsynbio/proteinlmdataset"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/set-based-prompting-provably-solving-the","slug":"set-based-prompting-provably-solving-the","title":"Order-Independence Without Fine Tuning","date":"2024-06-04","arxiv_id":"2406.06581","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":1,"n_instrument":7,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/set-based-prompting-provably-solving-the#ran","syntology_url":"https://syntology.ai/paper/2406.06581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06581"}},"official":{"repos":["reidmcy/set-based-prompting"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-large-language-model-biases-in","slug":"evaluating-large-language-model-biases-in","title":"Evaluating Large Language Model Biases in Persona-Steered Generation","date":"2024-05-30","arxiv_id":"2405.20253","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-large-language-model-biases-in#ran","syntology_url":"https://syntology.ai/paper/2405.20253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20253"}},"official":{"repos":["andyjliu/persona-steered-generation-bias"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/eliciting-informative-text-evaluations-with","slug":"eliciting-informative-text-evaluations-with","title":"Eliciting Informative Text Evaluations with Large Language Models","date":"2024-05-23","arxiv_id":"2405.15077","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/eliciting-informative-text-evaluations-with#ran","syntology_url":"https://syntology.ai/paper/2405.15077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15077"}},"official":{"repos":["yx-lu/eliciting-informative-text-evaluations-with-large-language-models"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-volatility-for-out-of-distribution","slug":"trajectory-volatility-for-out-of-distribution","title":"Embedding Trajectory for Out-of-Distribution Detection in Mathematical Reasoning","date":"2024-05-22","arxiv_id":"2405.14039","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-volatility-for-out-of-distribution#ran","syntology_url":"https://syntology.ai/paper/2405.14039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14039"}},"official":{"repos":["alsace08/ood-math-reasoning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multiple-choice-questions-are-efficient-and","slug":"multiple-choice-questions-are-efficient-and","title":"Multiple-Choice Questions are Efficient and Robust LLM Evaluators","date":"2024-05-20","arxiv_id":"2405.11966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multiple-choice-questions-are-efficient-and#ran","syntology_url":"https://syntology.ai/paper/2405.11966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11966"}},"official":{"repos":["geralt-targaryen/mc-evaluation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/limited-ability-of-llms-to-simulate-human","slug":"limited-ability-of-llms-to-simulate-human","title":"Limited Ability of LLMs to Simulate Human Psychological Behaviours: a Psychometric Analysis","date":"2024-05-12","arxiv_id":"2405.07248","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/limited-ability-of-llms-to-simulate-human#ran","syntology_url":"https://syntology.ai/paper/2405.07248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07248"}},"official":{"repos":["nikbpetrov/llms-simulate-humans"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-large-language-models-understand","slug":"do-large-language-models-understand","title":"Do Large Language Models Understand Conversational Implicature -- A case study with a chinese sitcom","date":"2024-04-30","arxiv_id":"2404.19509","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/do-large-language-models-understand#ran","syntology_url":"https://syntology.ai/paper/2404.19509","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.19509"}},"official":{"repos":["sjtu-compling/llm-pragmatics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/seed-bench-2-plus-benchmarking-multimodal","slug":"seed-bench-2-plus-benchmarking-multimodal","title":"SEED-Bench-2-Plus: Benchmarking Multimodal Large Language Models with Text-Rich Visual Comprehension","date":"2024-04-25","arxiv_id":"2404.16790","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seed-bench-2-plus-benchmarking-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.16790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16790"}},"official":{"repos":["ailab-cvc/seed-bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/taxi-evaluating-categorical-knowledge-editing","slug":"taxi-evaluating-categorical-knowledge-editing","title":"TAXI: Evaluating Categorical Knowledge Editing for Language Models","date":"2024-04-23","arxiv_id":"2404.15004","repositories_listed":1,"syntology":{"n":17,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/taxi-evaluating-categorical-knowledge-editing#ran","syntology_url":"https://syntology.ai/paper/2404.15004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15004"}},"official":{"repos":["derekpowell/taxi"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/ma-lmm-memory-augmented-large-multimodal","slug":"ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","arxiv_id":"2404.05726","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ma-lmm-memory-augmented-large-multimodal#ran","syntology_url":"https://syntology.ai/paper/2404.05726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05726"}},"official":{"repos":["boheumd/MA-LMM"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt4-video-advancing-multimodal-llms-for","slug":"minigpt4-video-advancing-multimodal-llms-for","title":"MiniGPT4-Video: Advancing Multimodal LLMs for Video Understanding with Interleaved Visual-Textual Tokens","date":"2024-04-04","arxiv_id":"2404.03413","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt4-video-advancing-multimodal-llms-for#ran","syntology_url":"https://syntology.ai/paper/2404.03413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03413"}},"official":{"repos":["Vision-CAIR/MiniGPT4-video"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-image-grid-can-be-worth-a-video-zero-shot","slug":"an-image-grid-can-be-worth-a-video-zero-shot","title":"An Image Grid Can Be Worth a Video: Zero-shot Video Question Answering Using a VLM","date":"2024-03-27","arxiv_id":"2403.18406","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-image-grid-can-be-worth-a-video-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2403.18406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18406"}},"official":{"repos":["imagegridworth/IG-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedlm-a-2-7b-parameter-language-model","slug":"biomedlm-a-2-7b-parameter-language-model","title":"BioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text","date":"2024-03-27","arxiv_id":"2403.18421","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/biomedlm-a-2-7b-parameter-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.18421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18421"}},"official":{"repos":["stanford-crfm/biomedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-long-videos-in-one-multimodal","slug":"understanding-long-videos-in-one-multimodal","title":"Understanding Long Videos with Multimodal Language Models","date":"2024-03-25","arxiv_id":"2403.16998","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-long-videos-in-one-multimodal#ran","syntology_url":"https://syntology.ai/paper/2403.16998","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16998"}},"official":{"repos":["kahnchana/mvu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/illusionvqa-a-challenging-optical-illusion","slug":"illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","arxiv_id":"2403.15952","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/illusionvqa-a-challenging-optical-illusion#ran","syntology_url":"https://syntology.ai/paper/2403.15952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15952"}},"official":{"repos":["csebuetnlp/illusionvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/complex-reasoning-over-logical-queries-on","slug":"complex-reasoning-over-logical-queries-on","title":"Complex Reasoning over Logical Queries on Commonsense Knowledge Graphs","date":"2024-03-12","arxiv_id":"2403.07398","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/complex-reasoning-over-logical-queries-on#ran","syntology_url":"https://syntology.ai/paper/2403.07398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07398"}},"official":{"repos":["tqfang/complex-commonsense-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/unfamiliar-finetuning-examples-control-how","slug":"unfamiliar-finetuning-examples-control-how","title":"Unfamiliar Finetuning Examples Control How Language Models Hallucinate","date":"2024-03-08","arxiv_id":"2403.05612","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unfamiliar-finetuning-examples-control-how#ran","syntology_url":"https://syntology.ai/paper/2403.05612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05612"}},"official":{"repos":["katiekang1998/llm_hallucinations"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-wmdp-benchmark-measuring-and-reducing","slug":"the-wmdp-benchmark-measuring-and-reducing","title":"The WMDP Benchmark: Measuring and Reducing Malicious Use With Unlearning","date":"2024-03-05","arxiv_id":"2403.03218","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-wmdp-benchmark-measuring-and-reducing#ran","syntology_url":"https://syntology.ai/paper/2403.03218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03218"}},"official":null}},{"url":"/paper/parallelparc-a-scalable-pipeline-for","slug":"parallelparc-a-scalable-pipeline-for","title":"ParallelPARC: A Scalable Pipeline for Generating Natural-Language Analogies","date":"2024-03-02","arxiv_id":"2403.01139","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallelparc-a-scalable-pipeline-for#ran","syntology_url":"https://syntology.ai/paper/2403.01139","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01139"}},"official":{"repos":["orensul/parallelparc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-2","slug":"benchmarking-large-language-models-on-2","title":"Benchmarking Large Language Models on Answering and Explaining Challenging Medical Questions","date":"2024-02-28","arxiv_id":"2402.18060","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-large-language-models-on-2#ran","syntology_url":"https://syntology.ai/paper/2402.18060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18060"}},"official":{"repos":["hanjiechen/challengeclinicalqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/political-compass-or-spinning-arrow-towards","slug":"political-compass-or-spinning-arrow-towards","title":"Political Compass or Spinning Arrow? Towards More Meaningful Evaluations for Values and Opinions in Large Language Models","date":"2024-02-26","arxiv_id":"2402.16786","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/political-compass-or-spinning-arrow-towards#ran","syntology_url":"https://syntology.ai/paper/2402.16786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16786"}},"official":{"repos":["paul-rottger/llm-values-pct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-large-language-models-for-learning","slug":"leveraging-large-language-models-for-learning","title":"Leveraging Large Language Models for Learning Complex Legal Concepts through Storytelling","date":"2024-02-26","arxiv_id":"2402.17019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/leveraging-large-language-models-for-learning#ran","syntology_url":"https://syntology.ai/paper/2402.17019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17019"}},"official":{"repos":["hjian42/legalstories"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tombench-benchmarking-theory-of-mind-in-large","slug":"tombench-benchmarking-theory-of-mind-in-large","title":"ToMBench: Benchmarking Theory of Mind in Large Language Models","date":"2024-02-23","arxiv_id":"2402.15052","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tombench-benchmarking-theory-of-mind-in-large#ran","syntology_url":"https://syntology.ai/paper/2402.15052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15052"}},"official":{"repos":["zhchen18/tombench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/tinybenchmarks-evaluating-llms-with-fewer","slug":"tinybenchmarks-evaluating-llms-with-fewer","title":"tinyBenchmarks: evaluating LLMs with fewer examples","date":"2024-02-22","arxiv_id":"2402.14992","repositories_listed":4,"syntology":{"n":39,"n_ran":25,"n_constructed":0,"n_ran_checked":25,"n_instrument":0,"n_unverified":14,"n_honours":0,"n_violates":1,"n_no_contract":24,"n_pointer_only":17,"phrase":"25 ran (of which 0 constructed an object rather than computing a result; 25 with no instrument failure: 0 honoured, 1 violated, 24 with no contract checked; 0 where Syntology's instrument failed) · 14 unverified","sample_list":"/paper/tinybenchmarks-evaluating-llms-with-fewer#ran","syntology_url":"https://syntology.ai/paper/2402.14992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14992"}},"official":{"repos":["felipemaiapolo/efficbench","felipemaiapolo/tinybenchmarks"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":12,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/arabicmmlu-assessing-massive-multitask","slug":"arabicmmlu-assessing-massive-multitask","title":"ArabicMMLU: Assessing Massive Multitask Language Understanding in Arabic","date":"2024-02-20","arxiv_id":"2402.12840","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/arabicmmlu-assessing-massive-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.12840","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12840"}},"official":{"repos":["mbzuai-nlp/arabicmmlu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/softmax-probabilities-mostly-predict-large","slug":"softmax-probabilities-mostly-predict-large","title":"Probabilities of Chat LLMs Are Miscalibrated but Still Predict Correctness on Multiple-Choice Q&A","date":"2024-02-20","arxiv_id":"2402.13213","repositories_listed":2,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/softmax-probabilities-mostly-predict-large#ran","syntology_url":"https://syntology.ai/paper/2402.13213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13213"}},"official":{"repos":["bplaut/softmax-probs-predict-llm-correctness","bplaut/llm-calibration-and-correctness-prediction"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-quantification-in-fine-tuned-llms","slug":"uncertainty-quantification-in-fine-tuned-llms","title":"Uncertainty quantification in fine-tuned LLMs using LoRA ensembles","date":"2024-02-19","arxiv_id":"2402.12264","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncertainty-quantification-in-fine-tuned-llms#ran","syntology_url":"https://syntology.ai/paper/2402.12264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12264"}},"official":{"repos":["oleksandr-balabanov/equivariant-posteriors"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/artifacts-or-abduction-how-do-llms-answer","slug":"artifacts-or-abduction-how-do-llms-answer","title":"Artifacts or Abduction: How Do LLMs Answer Multiple-Choice Questions Without the Question?","date":"2024-02-19","arxiv_id":"2402.12483","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/artifacts-or-abduction-how-do-llms-answer#ran","syntology_url":"https://syntology.ai/paper/2402.12483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12483"}},"official":{"repos":["nbalepur/mcqa-artifacts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/de-cop-detecting-copyrighted-content-in","slug":"de-cop-detecting-copyrighted-content-in","title":"DE-COP: Detecting Copyrighted Content in Language Models Training Data","date":"2024-02-15","arxiv_id":"2402.09910","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/de-cop-detecting-copyrighted-content-in#ran","syntology_url":"https://syntology.ai/paper/2402.09910","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09910"}},"official":{"repos":["avduarte333/de-cop_method","leililab/de-cop"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/salad-bench-a-hierarchical-and-comprehensive","slug":"salad-bench-a-hierarchical-and-comprehensive","title":"SALAD-Bench: A Hierarchical and Comprehensive Safety Benchmark for Large Language Models","date":"2024-02-07","arxiv_id":"2402.05044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/salad-bench-a-hierarchical-and-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.05044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05044"}},"official":{"repos":["opensafetylab/salad-bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-effect-of-sampling-temperature-on-problem","slug":"the-effect-of-sampling-temperature-on-problem","title":"The Effect of Sampling Temperature on Problem Solving in Large Language Models","date":"2024-02-07","arxiv_id":"2402.05201","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-effect-of-sampling-temperature-on-problem#ran","syntology_url":"https://syntology.ai/paper/2402.05201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05201"}},"official":{"repos":["matthewrenze/jhu-llm-temperature"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shield-an-evaluation-benchmark-for-face","slug":"shield-an-evaluation-benchmark-for-face","title":"SHIELD : An Evaluation Benchmark for Face Spoofing and Forgery Detection with Multimodal Large Language Models","date":"2024-02-06","arxiv_id":"2402.04178","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shield-an-evaluation-benchmark-for-face#ran","syntology_url":"https://syntology.ai/paper/2402.04178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04178"}},"official":{"repos":["laiyingxin2/shield"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-benchmarks-are-targets-revealing-the","slug":"when-benchmarks-are-targets-revealing-the","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","date":"2024-02-01","arxiv_id":"2402.01781","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-benchmarks-are-targets-revealing-the#ran","syntology_url":"https://syntology.ai/paper/2402.01781","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01781"}},"official":{"repos":["national-center-for-ai-saudi-arabia/lm-evaluation-harness"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/i-think-therefore-i-am-awareness-in-large","slug":"i-think-therefore-i-am-awareness-in-large","title":"I Think, Therefore I am: Benchmarking Awareness of Large Language Models Using AwareBench","date":"2024-01-31","arxiv_id":"2401.17882","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/i-think-therefore-i-am-awareness-in-large#ran","syntology_url":"https://syntology.ai/paper/2401.17882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17882"}},"official":{"repos":["howiehwong/awareness-in-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/e-eval-a-comprehensive-chinese-k-12-education","slug":"e-eval-a-comprehensive-chinese-k-12-education","title":"E-EVAL: A Comprehensive Chinese K-12 Education Evaluation Benchmark for Large Language Models","date":"2024-01-29","arxiv_id":"2401.15927","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/e-eval-a-comprehensive-chinese-k-12-education#ran","syntology_url":"https://syntology.ai/paper/2401.15927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15927"}},"official":{"repos":["ai-edu-lab/e-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cmmu-a-benchmark-for-chinese-multi-modal","slug":"cmmu-a-benchmark-for-chinese-multi-modal","title":"CMMU: A Benchmark for Chinese Multi-modal Multi-type Question Understanding and Reasoning","date":"2024-01-25","arxiv_id":"2401.14011","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cmmu-a-benchmark-for-chinese-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2401.14011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14011"}},"official":{"repos":["flagopen/cmmu"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longhealth-a-question-answering-benchmark","slug":"longhealth-a-question-answering-benchmark","title":"LongHealth: A Question Answering Benchmark with Long Clinical Documents","date":"2024-01-25","arxiv_id":"2401.14490","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longhealth-a-question-answering-benchmark#ran","syntology_url":"https://syntology.ai/paper/2401.14490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14490"}},"official":{"repos":["kbressem/longhealth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-external-knowledge-resources-to","slug":"leveraging-external-knowledge-resources-to","title":"Towards Efficient Methods in Medical Question Answering using Knowledge Graph Embeddings","date":"2024-01-15","arxiv_id":"2401.07977","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-external-knowledge-resources-to#ran","syntology_url":"https://syntology.ai/paper/2401.07977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07977"}},"official":{"repos":["saptarshi059/cdqa-project"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-benefits-of-a-concise-chain-of-thought-on","slug":"the-benefits-of-a-concise-chain-of-thought-on","title":"The Benefits of a Concise Chain of Thought on Problem-Solving in Large Language Models","date":"2024-01-11","arxiv_id":"2401.05618","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-benefits-of-a-concise-chain-of-thought-on#ran","syntology_url":"https://syntology.ai/paper/2401.05618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05618"}},"official":{"repos":["matthewrenze/jhu-concise-cot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/secqa-a-concise-question-answering-dataset#ran","syntology_url":"https://syntology.ai/paper/2312.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15838"}},"official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/an-in-depth-look-at-gemini-s-language#ran","syntology_url":"https://syntology.ai/paper/2312.11444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.11444"}},"official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/steering-llama-2-via-contrastive-activation","slug":"steering-llama-2-via-contrastive-activation","title":"Steering Llama 2 via Contrastive Activation Addition","date":"2023-12-09","arxiv_id":"2312.06681","repositories_listed":4,"syntology":{"n":22,"n_ran":21,"n_constructed":0,"n_ran_checked":21,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":19,"n_pointer_only":10,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 2 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/steering-llama-2-via-contrastive-activation#ran","syntology_url":"https://syntology.ai/paper/2312.06681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06681"}},"official":{"repos":["nrimsky/caa","nrimsky/sycophancysteering","wusche1/caa_hallucination"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/beyond-surface-probing-llama-across-scales","slug":"beyond-surface-probing-llama-across-scales","title":"Is Bigger and Deeper Always Better? Probing LLaMA Across Scales and Layers","date":"2023-12-07","arxiv_id":"2312.04333","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-surface-probing-llama-across-scales#ran","syntology_url":"https://syntology.ai/paper/2312.04333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04333"}},"official":{"repos":["nuochenpku/llama_analysis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"99c0d87b985df020555aa8ceccf13e116ab448eacaf9905ea6797d8dbdfe2d55","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}