{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multiple-choice/papers/ran/2","list_of":"/task/multiple-choice","task":"Multiple-choice","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,161],"of":161,"counts":{"archive_papers_tagged":1107,"with_a_code_link":483,"where_syntology_ran_a_sample":161,"not_listed_spam_title":0,"listed":1107,"listed_where_code_ran":161,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":124,"every_run_a_failure_of_syntologys_instrument":37,"listed_with_a_run_with_no_instrument_failure":124,"listed_every_run_a_failure_of_syntologys_instrument":37,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multiple-choice/papers/ran/1","prev":"/task/multiple-choice/papers/ran/1","next":null,"papers":[{"url":"/paper/clomo-counterfactual-logical-modification","slug":"clomo-counterfactual-logical-modification","title":"CLOMO: Counterfactual Logical Modification with Large Language Models","date":"2023-11-29","arxiv_id":"2311.17438","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/clomo-counterfactual-logical-modification#ran","syntology_url":"https://syntology.ai/paper/2311.17438","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17438"}},"official":{"repos":["eleanor-h/clomo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mvbench-a-comprehensive-multi-modal-video","slug":"mvbench-a-comprehensive-multi-modal-video","title":"MVBench: A Comprehensive Multi-modal Video Understanding Benchmark","date":"2023-11-28","arxiv_id":"2311.17005","repositories_listed":3,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mvbench-a-comprehensive-multi-modal-video#ran","syntology_url":"https://syntology.ai/paper/2311.17005","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17005"}},"official":{"repos":["opengvlab/ask-anything"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gpqa-a-graduate-level-google-proof-q-a","slug":"gpqa-a-graduate-level-google-proof-q-a","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","date":"2023-11-20","arxiv_id":"2311.12022","repositories_listed":3,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gpqa-a-graduate-level-google-proof-q-a#ran","syntology_url":"https://syntology.ai/paper/2311.12022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.12022"}},"official":{"repos":["idavidrein/gpqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/performance-trade-offs-of-watermarking-large","slug":"performance-trade-offs-of-watermarking-large","title":"Downstream Trade-offs of a Family of Text Watermarks","date":"2023-11-16","arxiv_id":"2311.09816","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/performance-trade-offs-of-watermarking-large#ran","syntology_url":"https://syntology.ai/paper/2311.09816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09816"}},"official":{"repos":["flair-iisc/watermark_tradeoffs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","repositories_listed":6,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}},"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/data-contamination-quiz-a-tool-to-detect-and","slug":"data-contamination-quiz-a-tool-to-detect-and","title":"Data Contamination Quiz: A Tool to Detect and Estimate Contamination in Large Language Models","date":"2023-11-10","arxiv_id":"2311.06233","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/data-contamination-quiz-a-tool-to-detect-and#ran","syntology_url":"https://syntology.ai/paper/2311.06233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06233"}},"official":{"repos":["shahriargolchin/dcq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/resilient-multiple-choice-learning-a-learned","slug":"resilient-multiple-choice-learning-a-learned","title":"Resilient Multiple Choice Learning: A learned scoring scheme with application to audio scene analysis","date":"2023-11-02","arxiv_id":"2311.01052","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/resilient-multiple-choice-learning-a-learned#ran","syntology_url":"https://syntology.ai/paper/2311.01052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01052"}},"official":{"repos":["victorletzelter/code-rmcl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/brainteaser-lateral-thinking-puzzles-for","slug":"brainteaser-lateral-thinking-puzzles-for","title":"BRAINTEASER: Lateral Thinking Puzzles for Large Language Models","date":"2023-10-08","arxiv_id":"2310.05057","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/brainteaser-lateral-thinking-puzzles-for#ran","syntology_url":"https://syntology.ai/paper/2310.05057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05057"}},"official":null}},{"url":"/paper/evaluating-multi-agent-coordination-abilities","slug":"evaluating-multi-agent-coordination-abilities","title":"LLM-Coordination: Evaluating and Analyzing Multi-agent Coordination Abilities in Large Language Models","date":"2023-10-05","arxiv_id":"2310.03903","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-multi-agent-coordination-abilities#ran","syntology_url":"https://syntology.ai/paper/2310.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03903"}},"official":{"repos":["eric-ai-lab/llm_coordination"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/autocast-enhancing-world-event-prediction","slug":"autocast-enhancing-world-event-prediction","title":"AutoCast++: Enhancing World Event Prediction with Zero-shot Ranking-based Context Retrieval","date":"2023-10-03","arxiv_id":"2310.01880","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/autocast-enhancing-world-event-prediction#ran","syntology_url":"https://syntology.ai/paper/2310.01880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01880"}},"official":{"repos":["BorealisAI/Autocast-plus-plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-as-knowledge-bases-for-visual","slug":"language-models-as-knowledge-bases-for-visual","title":"Language Models as Knowledge Bases for Visual Word Sense Disambiguation","date":"2023-10-03","arxiv_id":"2310.01960","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-models-as-knowledge-bases-for-visual#ran","syntology_url":"https://syntology.ai/paper/2310.01960","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01960"}},"official":{"repos":["anastasiakrith/llm-for-vwsd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fusing-models-with-complementary-expertise","slug":"fusing-models-with-complementary-expertise","title":"Fusing Models with Complementary Expertise","date":"2023-10-02","arxiv_id":"2310.01542","repositories_listed":1,"syntology":{"n":19,"n_ran":13,"n_constructed":2,"n_ran_checked":6,"n_instrument":7,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":19,"phrase":"13 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 7 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/fusing-models-with-complementary-expertise#ran","syntology_url":"https://syntology.ai/paper/2310.01542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01542"}},"official":{"repos":["hwang595/foe-iclr2024"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":2,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/fool-your-vision-and-language-model-with","slug":"fool-your-vision-and-language-model-with","title":"Fool Your (Vision and) Language Model With Embarrassingly Simple Permutations","date":"2023-10-02","arxiv_id":"2310.01651","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/fool-your-vision-and-language-model-with#ran","syntology_url":"https://syntology.ai/paper/2310.01651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01651"}},"official":{"repos":["ys-zong/foolyourvllms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/exploring-self-reinforcement-for-improving","slug":"exploring-self-reinforcement-for-improving","title":"Exploring Iterative Enhancement for Improving Learnersourced Multiple-Choice Question Explanations with Large Language Models","date":"2023-09-19","arxiv_id":"2309.10444","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exploring-self-reinforcement-for-improving#ran","syntology_url":"https://syntology.ai/paper/2309.10444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.10444"}},"official":{"repos":["strong-ai-lab/explanation-generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/safetybench-evaluating-the-safety-of-large","slug":"safetybench-evaluating-the-safety-of-large","title":"SafetyBench: Evaluating the Safety of Large Language Models","date":"2023-09-13","arxiv_id":"2309.07045","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safetybench-evaluating-the-safety-of-large#ran","syntology_url":"https://syntology.ai/paper/2309.07045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07045"}},"official":{"repos":["thu-coai/safetybench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/on-large-language-models-selection-bias-in","slug":"on-large-language-models-selection-bias-in","title":"Large Language Models Are Not Robust Multiple Choice Selectors","date":"2023-09-07","arxiv_id":"2309.03882","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-large-language-models-selection-bias-in#ran","syntology_url":"https://syntology.ai/paper/2309.03882","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.03882"}},"official":{"repos":["chujiezheng/llm-mcq-bias"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fineval-a-chinese-financial-domain-knowledge","slug":"fineval-a-chinese-financial-domain-knowledge","title":"FinEval: A Chinese Financial Domain Knowledge Evaluation Benchmark for Large Language Models","date":"2023-08-19","arxiv_id":"2308.09975","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fineval-a-chinese-financial-domain-knowledge#ran","syntology_url":"https://syntology.ai/paper/2308.09975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09975"}},"official":{"repos":["sufe-aiflm-lab/fineval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/open-vocabulary-video-question-answering-a","slug":"open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","arxiv_id":"2308.09363","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-vocabulary-video-question-answering-a#ran","syntology_url":"https://syntology.ai/paper/2308.09363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09363"}},"official":{"repos":["mlvlab/ovqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/egoschema-a-diagnostic-benchmark-for-very-1","slug":"egoschema-a-diagnostic-benchmark-for-very-1","title":"EgoSchema: A Diagnostic Benchmark for Very Long-form Video Language Understanding","date":"2023-08-17","arxiv_id":"2308.09126","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/egoschema-a-diagnostic-benchmark-for-very-1#ran","syntology_url":"https://syntology.ai/paper/2308.09126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09126"}},"official":{"repos":["egoschema/egoschema"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-human-like-multi-modal-reasoning-a","slug":"enhancing-human-like-multi-modal-reasoning-a","title":"Enhancing Human-like Multi-Modal Reasoning: A New Challenging Dataset and Comprehensive Framework","date":"2023-07-24","arxiv_id":"2307.12626","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-human-like-multi-modal-reasoning-a#ran","syntology_url":"https://syntology.ai/paper/2307.12626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12626"}},"official":{"repos":["weijingxuan/COCO-MMR"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/scibench-evaluating-college-level-scientific","slug":"scibench-evaluating-college-level-scientific","title":"SciBench: Evaluating College-Level Scientific Problem-Solving Abilities of Large Language Models","date":"2023-07-20","arxiv_id":"2307.10635","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scibench-evaluating-college-level-scientific#ran","syntology_url":"https://syntology.ai/paper/2307.10635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.10635"}},"official":{"repos":["mandyyyyii/scibench"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/questioning-the-survey-responses-of-large","slug":"questioning-the-survey-responses-of-large","title":"Questioning the Survey Responses of Large Language Models","date":"2023-06-13","arxiv_id":"2306.07951","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/questioning-the-survey-responses-of-large#ran","syntology_url":"https://syntology.ai/paper/2306.07951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07951"}},"official":{"repos":["socialfoundations/surveying-language-models"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-cmexam","slug":"benchmarking-large-language-models-on-cmexam","title":"Benchmarking Large Language Models on CMExam -- A Comprehensive Chinese Medical Exam Dataset","date":"2023-06-05","arxiv_id":"2306.03030","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-large-language-models-on-cmexam#ran","syntology_url":"https://syntology.ai/paper/2306.03030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.03030"}},"official":{"repos":["williamliujl/cmexam"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conformal-prediction-with-large-language","slug":"conformal-prediction-with-large-language","title":"Conformal Prediction with Large Language Models for Multi-Choice Question Answering","date":"2023-05-28","arxiv_id":"2305.18404","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/conformal-prediction-with-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.18404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18404"}},"official":{"repos":["bhaweshiitk/conformalllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-language-models-with-just-forward-1","slug":"fine-tuning-language-models-with-just-forward-1","title":"Fine-Tuning Language Models with Just Forward Passes","date":"2023-05-27","arxiv_id":"2305.17333","repositories_listed":3,"syntology":{"n":17,"n_ran":10,"n_constructed":2,"n_ran_checked":8,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"10 ran (of which 2 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/fine-tuning-language-models-with-just-forward-1#ran","syntology_url":"https://syntology.ai/paper/2305.17333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17333"}},"official":{"repos":["princeton-nlp/mezo"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/c-eval-a-multi-level-multi-discipline-chinese-1","slug":"c-eval-a-multi-level-multi-discipline-chinese-1","title":"C-Eval: A Multi-Level Multi-Discipline Chinese Evaluation Suite for Foundation Models","date":"2023-05-15","arxiv_id":"2305.08322","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/c-eval-a-multi-level-multi-discipline-chinese-1#ran","syntology_url":"https://syntology.ai/paper/2305.08322","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.08322"}},"official":{"repos":["hkust-nlp/ceval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-don-t-always-say-what-they-1","slug":"language-models-don-t-always-say-what-they-1","title":"Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting","date":"2023-05-07","arxiv_id":"2305.04388","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-models-don-t-always-say-what-they-1#ran","syntology_url":"https://syntology.ai/paper/2305.04388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04388"}},"official":{"repos":["milesaturpin/cot-unfaithfulness"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-gpt-3-5-and-gpt-4-models-on","slug":"evaluating-gpt-3-5-and-gpt-4-models-on","title":"Evaluating GPT-3.5 and GPT-4 Models on Brazilian University Admission Exams","date":"2023-03-29","arxiv_id":"2303.17003","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-gpt-3-5-and-gpt-4-models-on#ran","syntology_url":"https://syntology.ai/paper/2303.17003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.17003"}},"official":{"repos":["piresramon/gpt-4-enem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explicit-planning-helps-language-models-in","slug":"explicit-planning-helps-language-models-in","title":"Explicit Planning Helps Language Models in Logical Reasoning","date":"2023-03-28","arxiv_id":"2303.15714","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/explicit-planning-helps-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2303.15714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.15714"}},"official":{"repos":["cindermond/explicit-planning-for-reasoning","cindermond/leap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mqag-multiple-choice-question-answering-and","slug":"mqag-multiple-choice-question-answering-and","title":"MQAG: Multiple-choice Question Answering and Generation for Assessing Information Consistency in Summarization","date":"2023-01-28","arxiv_id":"2301.12307","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mqag-multiple-choice-question-answering-and#ran","syntology_url":"https://syntology.ai/paper/2301.12307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12307"}},"official":{"repos":["potsawee/mqag0"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-large-language-models-for-multiple","slug":"leveraging-large-language-models-for-multiple","title":"Leveraging Large Language Models for Multiple Choice Question Answering","date":"2022-10-22","arxiv_id":"2210.12353","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-large-language-models-for-multiple#ran","syntology_url":"https://syntology.ai/paper/2210.12353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.12353"}},"official":{"repos":["byu-pccl/leveraging-llms-for-mcqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-open-domain-question-answering","slug":"variational-open-domain-question-answering","title":"Variational Open-Domain Question Answering","date":"2022-09-23","arxiv_id":"2210.06345","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/variational-open-domain-question-answering#ran","syntology_url":"https://syntology.ai/paper/2210.06345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06345"}},"official":{"repos":["findzebra/fz-openqa","VodLM/vod"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learn-to-explain-multimodal-reasoning-via","slug":"learn-to-explain-multimodal-reasoning-via","title":"Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering","date":"2022-09-20","arxiv_id":"2209.09513","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learn-to-explain-multimodal-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2209.09513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.09513"}},"official":{"repos":["lupantech/ScienceQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/can-large-language-models-reason-about","slug":"can-large-language-models-reason-about","title":"Can large language models reason about medical questions?","date":"2022-07-17","arxiv_id":"2207.08143","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-large-language-models-reason-about#ran","syntology_url":"https://syntology.ai/paper/2207.08143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08143"}},"official":{"repos":["vlievin/medical-reasoning"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/feta-a-benchmark-for-few-sample-task-transfer","slug":"feta-a-benchmark-for-few-sample-task-transfer","title":"FETA: A Benchmark for Few-Sample Task Transfer in Open-Domain Dialogue","date":"2022-05-12","arxiv_id":"2205.06262","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/feta-a-benchmark-for-few-sample-task-transfer#ran","syntology_url":"https://syntology.ai/paper/2205.06262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.06262"}},"official":{"repos":["alon-albalak/tlidb"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clues-before-answers-generation-enhanced","slug":"clues-before-answers-generation-enhanced","title":"Clues Before Answers: Generation-Enhanced Multiple-Choice QA","date":"2022-04-30","arxiv_id":"2205.00274","repositories_listed":1,"syntology":{"n":11,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/clues-before-answers-generation-enhanced#ran","syntology_url":"https://syntology.ai/paper/2205.00274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.00274"}},"official":{"repos":["nju-websoft/genmc"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","repositories_listed":5,"syntology":{"n":24,"n_ran":18,"n_constructed":6,"n_ran_checked":12,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":8,"phrase":"18 ran (of which 6 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}},"official":null}},{"url":"/paper/adalogn-adaptive-logic-graph-network-for","slug":"adalogn-adaptive-logic-graph-network-for","title":"AdaLoGN: Adaptive Logic Graph Network for Reasoning-Based Machine Reading Comprehension","date":"2022-03-16","arxiv_id":"2203.08992","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adalogn-adaptive-logic-graph-network-for#ran","syntology_url":"https://syntology.ai/paper/2203.08992","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.08992"}},"official":{"repos":["nju-websoft/adalogn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bridgeformer-bridging-video-text-retrieval","slug":"bridgeformer-bridging-video-text-retrieval","title":"Bridging Video-text Retrieval with Multiple Choice Questions","date":"2022-01-13","arxiv_id":"2201.04850","repositories_listed":2,"syntology":{"n":24,"n_ran":13,"n_constructed":8,"n_ran_checked":9,"n_instrument":4,"n_unverified":11,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":6,"phrase":"13 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/bridgeformer-bridging-video-text-retrieval#ran","syntology_url":"https://syntology.ai/paper/2201.04850","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.04850"}},"official":{"repos":["tencentarc/mcq"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-few-more-examples-may-be-worth-billions-of","slug":"a-few-more-examples-may-be-worth-billions-of","title":"A Few More Examples May Be Worth Billions of Parameters","date":"2021-10-08","arxiv_id":"2110.04374","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":6,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 2 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-few-more-examples-may-be-worth-billions-of#ran","syntology_url":"https://syntology.ai/paper/2110.04374","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04374"}},"official":{"repos":["yuvalkirstain/lm-evaluation-harness"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prost-physical-reasoning-of-objects-through","slug":"prost-physical-reasoning-of-objects-through","title":"PROST: Physical Reasoning of Objects through Space and Time","date":"2021-06-07","arxiv_id":"2106.03634","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prost-physical-reasoning-of-objects-through#ran","syntology_url":"https://syntology.ai/paper/2106.03634","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.03634"}},"official":{"repos":["nala-cub/prost"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/neuralwoz-learning-to-collect-task-oriented","slug":"neuralwoz-learning-to-collect-task-oriented","title":"NeuralWOZ: Learning to Collect Task-Oriented Dialogue via Model-Based Simulation","date":"2021-05-30","arxiv_id":"2105.14454","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neuralwoz-learning-to-collect-task-oriented#ran","syntology_url":"https://syntology.ai/paper/2105.14454","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.14454"}},"official":{"repos":["naver-ai/neuralwoz"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/when-does-pretraining-help-assessing-self","slug":"when-does-pretraining-help-assessing-self","title":"When Does Pretraining Help? Assessing Self-Supervised Learning for Law and the CaseHOLD Dataset","date":"2021-04-18","arxiv_id":"2104.08671","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/when-does-pretraining-help-assessing-self#ran","syntology_url":"https://syntology.ai/paper/2104.08671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08671"}},"official":{"repos":["reglab/casehold"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/what-to-pre-train-on-efficient-intermediate","slug":"what-to-pre-train-on-efficient-intermediate","title":"What to Pre-Train on? Efficient Intermediate Task Selection","date":"2021-04-16","arxiv_id":"2104.08247","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/what-to-pre-train-on-efficient-intermediate#ran","syntology_url":"https://syntology.ai/paper/2104.08247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08247"}},"official":{"repos":["adapter-hub/efficient-task-transfer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/surface-form-competition-why-the-highest","slug":"surface-form-competition-why-the-highest","title":"Surface Form Competition: Why the Highest Probability Answer Isn't Always Right","date":"2021-04-16","arxiv_id":"2104.08315","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surface-form-competition-why-the-highest#ran","syntology_url":"https://syntology.ai/paper/2104.08315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08315"}},"official":{"repos":["peterwestuw/surface-form-competition"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/explagraphs-an-explanation-graph-generation","slug":"explagraphs-an-explanation-graph-generation","title":"ExplaGraphs: An Explanation Graph Generation Task for Structured Commonsense Reasoning","date":"2021-04-15","arxiv_id":"2104.07644","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/explagraphs-an-explanation-graph-generation#ran","syntology_url":"https://syntology.ai/paper/2104.07644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.07644"}},"official":{"repos":["swarnaHub/ExplaGraphs"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/trajectory-wise-multiple-choice-learning-for","slug":"trajectory-wise-multiple-choice-learning-for","title":"Trajectory-wise Multiple Choice Learning for Dynamics Generalization in Reinforcement Learning","date":"2020-10-26","arxiv_id":"2010.13303","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trajectory-wise-multiple-choice-learning-for#ran","syntology_url":"https://syntology.ai/paper/2010.13303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.13303"}},"official":{"repos":["younggyoseo/trajectory_mcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-disease-does-this-patient-have-a-large","slug":"what-disease-does-this-patient-have-a-large","title":"What Disease does this Patient Have? A Large-scale Open Domain Question Answering Dataset from Medical Exams","date":"2020-09-28","arxiv_id":"2009.13081","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-disease-does-this-patient-have-a-large#ran","syntology_url":"https://syntology.ai/paper/2009.13081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.13081"}},"official":{"repos":["jind11/MedQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unifiedqa-crossing-format-boundaries-with-a","slug":"unifiedqa-crossing-format-boundaries-with-a","title":"UnifiedQA: Crossing Format Boundaries With a Single QA System","date":"2020-05-02","arxiv_id":"2005.00700","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unifiedqa-crossing-format-boundaries-with-a#ran","syntology_url":"https://syntology.ai/paper/2005.00700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.00700"}},"official":{"repos":["allenai/unifiedqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/starc-structured-annotations-for-reading","slug":"starc-structured-annotations-for-reading","title":"STARC: Structured Annotations for Reading Comprehension","date":"2020-04-30","arxiv_id":"2004.14797","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/starc-structured-annotations-for-reading#ran","syntology_url":"https://syntology.ai/paper/2004.14797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14797"}},"official":{"repos":["berzak/onestop-qa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/logic-guided-data-augmentation-and","slug":"logic-guided-data-augmentation-and","title":"Logic-Guided Data Augmentation and Regularization for Consistent Question Answering","date":"2020-04-21","arxiv_id":"2004.10157","repositories_listed":1,"syntology":{"n":18,"n_ran":8,"n_constructed":0,"n_ran_checked":0,"n_instrument":8,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 8 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/logic-guided-data-augmentation-and#ran","syntology_url":"https://syntology.ai/paper/2004.10157","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.10157"}},"official":{"repos":["AkariAsai/logic_guided_qa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-commonsense-question-answering","slug":"unsupervised-commonsense-question-answering","title":"Unsupervised Commonsense Question Answering with Self-Talk","date":"2020-04-11","arxiv_id":"2004.05483","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/unsupervised-commonsense-question-answering#ran","syntology_url":"https://syntology.ai/paper/2004.05483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.05483"}},"official":{"repos":["vered1986/self_talk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/dmcl-distillation-multiple-choice-learning","slug":"dmcl-distillation-multiple-choice-learning","title":"DMCL: Distillation Multiple Choice Learning for Multimodal Action Recognition","date":"2019-12-23","arxiv_id":"1912.10982","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dmcl-distillation-multiple-choice-learning#ran","syntology_url":"https://syntology.ai/paper/1912.10982","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.10982"}},"official":{"repos":["ncgarcia/DMCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/abductive-commonsense-reasoning","slug":"abductive-commonsense-reasoning","title":"Abductive Commonsense Reasoning","date":"2019-08-15","arxiv_id":"1908.05739","repositories_listed":2,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":2,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/abductive-commonsense-reasoning#ran","syntology_url":"https://syntology.ai/paper/1908.05739","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1908.05739"}},"official":null}},{"url":"/paper/socialiqa-commonsense-reasoning-about-social","slug":"socialiqa-commonsense-reasoning-about-social","title":"SocialIQA: Commonsense Reasoning about Social Interactions","date":"2019-04-22","arxiv_id":"1904.09728","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/socialiqa-commonsense-reasoning-about-social#ran","syntology_url":"https://syntology.ai/paper/1904.09728","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.09728"}},"official":null}},{"url":"/paper/from-recognition-to-cognition-visual","slug":"from-recognition-to-cognition-visual","title":"From Recognition to Cognition: Visual Commonsense Reasoning","date":"2018-11-27","arxiv_id":"1811.10830","repositories_listed":4,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/from-recognition-to-cognition-visual#ran","syntology_url":"https://syntology.ai/paper/1811.10830","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.10830"}},"official":null}},{"url":"/paper/commonsenseqa-a-question-answering-challenge","slug":"commonsenseqa-a-question-answering-challenge","title":"CommonsenseQA: A Question Answering Challenge Targeting Commonsense Knowledge","date":"2018-11-02","arxiv_id":"1811.00937","repositories_listed":4,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/commonsenseqa-a-question-answering-challenge#ran","syntology_url":"https://syntology.ai/paper/1811.00937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.00937"}},"official":{"repos":["jonathanherzig/commonsenseqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/a-joint-sequence-fusion-model-for-video","slug":"a-joint-sequence-fusion-model-for-video","title":"A Joint Sequence Fusion Model for Video Question Answering and Retrieval","date":"2018-08-07","arxiv_id":"1808.02559","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-joint-sequence-fusion-model-for-video#ran","syntology_url":"https://syntology.ai/paper/1808.02559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.02559"}},"official":null}},{"url":"/paper/a-simple-method-for-commonsense-reasoning","slug":"a-simple-method-for-commonsense-reasoning","title":"A Simple Method for Commonsense Reasoning","date":"2018-06-07","arxiv_id":"1806.02847","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-simple-method-for-commonsense-reasoning#ran","syntology_url":"https://syntology.ai/paper/1806.02847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.02847"}},"official":null}},{"url":"/paper/vqa-visual-question-answering","slug":"vqa-visual-question-answering","title":"VQA: Visual Question Answering","date":"2015-05-03","arxiv_id":"1505.00468","repositories_listed":21,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vqa-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/1505.00468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1505.00468"}},"official":null}}],"record_sha256":"14eba80dcaee8d9435cf1bcea0fe4f72d4ab1955a17c44cce566f966c058926f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}