{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/ran/5","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":5,"pages_in_order":13,"rows_per_page":100,"rows":[401,500],"of":1274,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering/papers/ran/1","prev":"/task/question-answering/papers/ran/4","next":"/task/question-answering/papers/ran/6","papers":[{"url":"/paper/calibrating-large-language-models-using-their","slug":"calibrating-large-language-models-using-their","title":"Calibrating Large Language Models Using Their Generations Only","date":"2024-03-09","arxiv_id":"2403.05973","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calibrating-large-language-models-using-their#ran","syntology_url":"https://syntology.ai/paper/2403.05973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05973"}},"official":{"repos":["parameterlab/apricot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-t-remember-details-in-long-documents-you","slug":"can-t-remember-details-in-long-documents-you","title":"Can't Remember Details in Long Documents? You Need Some R&R","date":"2024-03-08","arxiv_id":"2403.05004","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-t-remember-details-in-long-documents-you#ran","syntology_url":"https://syntology.ai/paper/2403.05004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05004"}},"official":{"repos":["casetext/r-and-r"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/debiasing-large-visual-language-models","slug":"debiasing-large-visual-language-models","title":"Debiasing Multimodal Large Language Models","date":"2024-03-08","arxiv_id":"2403.05262","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/debiasing-large-visual-language-models#ran","syntology_url":"https://syntology.ai/paper/2403.05262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05262"}},"official":{"repos":["yfzhang114/llava-align"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/bias-augmented-consistency-training-reduces","slug":"bias-augmented-consistency-training-reduces","title":"Bias-Augmented Consistency Training Reduces Biased Reasoning in Chain-of-Thought","date":"2024-03-08","arxiv_id":"2403.05518","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bias-augmented-consistency-training-reduces#ran","syntology_url":"https://syntology.ai/paper/2403.05518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05518"}},"official":{"repos":["raybears/cot-transparency"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/textmonkey-an-ocr-free-large-multimodal-model","slug":"textmonkey-an-ocr-free-large-multimodal-model","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","date":"2024-03-07","arxiv_id":"2403.04473","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textmonkey-an-ocr-free-large-multimodal-model#ran","syntology_url":"https://syntology.ai/paper/2403.04473","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04473"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cat-enhancing-multimodal-large-language-model","slug":"cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","arxiv_id":"2403.04640","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cat-enhancing-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.04640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04640"}},"official":{"repos":["rikeilong/bay-cat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/few-shot-chain-of-thought-driven-reasoning-to","slug":"few-shot-chain-of-thought-driven-reasoning-to","title":"Few shot chain-of-thought driven reasoning to prompt LLMs for open ended medical question answering","date":"2024-03-07","arxiv_id":"2403.04890","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/few-shot-chain-of-thought-driven-reasoning-to#ran","syntology_url":"https://syntology.ai/paper/2403.04890","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04890"}},"official":{"repos":["coldseal/clinicr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-elementary-multilingual","slug":"evaluating-the-elementary-multilingual","title":"Evaluating the Elementary Multilingual Capabilities of Large Language Models with MultiQ","date":"2024-03-06","arxiv_id":"2403.03814","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-the-elementary-multilingual#ran","syntology_url":"https://syntology.ai/paper/2403.03814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03814"}},"official":{"repos":["paul-rottger/multiq"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/evidence-focused-fact-summarization-for","slug":"evidence-focused-fact-summarization-for","title":"Evidence-Focused Fact Summarization for Knowledge-Augmented Zero-Shot Question Answering","date":"2024-03-05","arxiv_id":"2403.02966","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":16,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/evidence-focused-fact-summarization-for#ran","syntology_url":"https://syntology.ai/paper/2403.02966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02966"}},"official":{"repos":["anon809/efsum"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/cr-lt-kgqa-a-knowledge-graph-question","slug":"cr-lt-kgqa-a-knowledge-graph-question","title":"CR-LT-KGQA: A Knowledge Graph Question Answering Dataset Requiring Commonsense Reasoning and Long-Tail Knowledge","date":"2024-03-03","arxiv_id":"2403.01395","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cr-lt-kgqa-a-knowledge-graph-question#ran","syntology_url":"https://syntology.ai/paper/2403.01395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01395"}},"official":{"repos":["d3mlab/cr-lt-kgqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-tuning-vs-retrieval-augmented-generation","slug":"fine-tuning-vs-retrieval-augmented-generation","title":"Fine Tuning vs. Retrieval Augmented Generation for Less Popular Knowledge","date":"2024-03-03","arxiv_id":"2403.01432","repositories_listed":1,"syntology":{"n":19,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":19,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/fine-tuning-vs-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2403.01432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01432"}},"official":{"repos":["heydarsoudani/ragvsft"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/let-llms-take-on-the-latest-challenges-a","slug":"let-llms-take-on-the-latest-challenges-a","title":"Let LLMs Take on the Latest Challenges! A Chinese Dynamic Question Answering Benchmark","date":"2024-02-29","arxiv_id":"2402.19248","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-llms-take-on-the-latest-challenges-a#ran","syntology_url":"https://syntology.ai/paper/2402.19248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19248"}},"official":{"repos":["alibaba-nlp/cdqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/characterizing-truthfulness-in-large-language","slug":"characterizing-truthfulness-in-large-language","title":"Characterizing Truthfulness in Large Language Model Generations with Local Intrinsic Dimension","date":"2024-02-28","arxiv_id":"2402.18048","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/characterizing-truthfulness-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.18048","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18048"}},"official":{"repos":["fanyin3639/lid-hallucinationdetection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-on-2","slug":"benchmarking-large-language-models-on-2","title":"Benchmarking Large Language Models on Answering and Explaining Challenging Medical Questions","date":"2024-02-28","arxiv_id":"2402.18060","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-large-language-models-on-2#ran","syntology_url":"https://syntology.ai/paper/2402.18060","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18060"}},"official":{"repos":["hanjiechen/challengeclinicalqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unsupervised-information-refinement-training","slug":"unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","arxiv_id":"2402.18150","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsupervised-information-refinement-training#ran","syntology_url":"https://syntology.ai/paper/2402.18150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18150"}},"official":{"repos":["xsc1234/info-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fact-and-reflection-far-improves-confidence","slug":"fact-and-reflection-far-improves-confidence","title":"Fact-and-Reflection (FaR) Improves Confidence Calibration of Large Language Models","date":"2024-02-27","arxiv_id":"2402.17124","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fact-and-reflection-far-improves-confidence#ran","syntology_url":"https://syntology.ai/paper/2402.17124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17124"}},"official":{"repos":["colinzhaoust/fact-and-reflection"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llm-generate-culturally-relevant","slug":"can-llm-generate-culturally-relevant","title":"Can LLM Generate Culturally Relevant Commonsense QA Data? Case Study in Indonesian and Sundanese","date":"2024-02-27","arxiv_id":"2402.17302","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-llm-generate-culturally-relevant#ran","syntology_url":"https://syntology.ai/paper/2402.17302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17302"}},"official":{"repos":["rifkiaputri/id-csqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rear-a-relevance-aware-retrieval-augmented","slug":"rear-a-relevance-aware-retrieval-augmented","title":"REAR: A Relevance-Aware Retrieval-Augmented Framework for Open-Domain Question Answering","date":"2024-02-27","arxiv_id":"2402.17497","repositories_listed":1,"syntology":{"n":14,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":14,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/rear-a-relevance-aware-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2402.17497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17497"}},"official":{"repos":["rucaibox/rear"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-very-long-term-conversational","slug":"evaluating-very-long-term-conversational","title":"Evaluating Very Long-Term Conversational Memory of LLM Agents","date":"2024-02-27","arxiv_id":"2402.17753","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/evaluating-very-long-term-conversational#ran","syntology_url":"https://syntology.ai/paper/2402.17753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17753"}},"official":null}},{"url":"/paper/truthx-alleviating-hallucinations-by-editing","slug":"truthx-alleviating-hallucinations-by-editing","title":"TruthX: Alleviating Hallucinations by Editing Large Language Models in Truthful Space","date":"2024-02-27","arxiv_id":"2402.17811","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/truthx-alleviating-hallucinations-by-editing#ran","syntology_url":"https://syntology.ai/paper/2402.17811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17811"}},"official":{"repos":["ictnlp/truthx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ehrnoteqa-a-patient-specific-question","slug":"ehrnoteqa-a-patient-specific-question","title":"EHRNoteQA: An LLM Benchmark for Real-World Clinical Practice Using Discharge Summaries","date":"2024-02-25","arxiv_id":"2402.16040","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ehrnoteqa-a-patient-specific-question#ran","syntology_url":"https://syntology.ai/paper/2402.16040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16040"}},"official":{"repos":["ji-youn-kim/ehrnoteqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lstp-language-guided-spatial-temporal-prompt","slug":"lstp-language-guided-spatial-temporal-prompt","title":"Efficient Temporal Extrapolation of Multimodal Large Language Models with Temporal Grounding Bridge","date":"2024-02-25","arxiv_id":"2402.16050","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lstp-language-guided-spatial-temporal-prompt#ran","syntology_url":"https://syntology.ai/paper/2402.16050","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16050"}},"official":{"repos":["bigai-nlco/lstp-chat","bigai-nlco/videotgb"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-gap-between-2d-and-3d-visual","slug":"bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","arxiv_id":"2402.15933","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-the-gap-between-2d-and-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2402.15933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15933"}},"official":{"repos":["matthewdm0816/bridgeqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-kbqa-multi-turn-interactions-for","slug":"interactive-kbqa-multi-turn-interactions-for","title":"Interactive-KBQA: Multi-Turn Interactions for Knowledge Base Question Answering with Large Language Models","date":"2024-02-23","arxiv_id":"2402.15131","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":3,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/interactive-kbqa-multi-turn-interactions-for#ran","syntology_url":"https://syntology.ai/paper/2402.15131","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15131"}},"official":{"repos":["jimxionggm/interactive-kbqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/triad-a-framework-leveraging-a-multi-role-llm","slug":"triad-a-framework-leveraging-a-multi-role-llm","title":"Triad: A Framework Leveraging a Multi-Role LLM-based Agent to Solve Knowledge Base Question Answering","date":"2024-02-22","arxiv_id":"2402.14320","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/triad-a-framework-leveraging-a-multi-role-llm#ran","syntology_url":"https://syntology.ai/paper/2402.14320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14320"}},"official":{"repos":["ZJU-DCDLab/Triad"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-hallucinations-of-multi-modal-large","slug":"visual-hallucinations-of-multi-modal-large","title":"Visual Hallucinations of Multi-modal Large Language Models","date":"2024-02-22","arxiv_id":"2402.14683","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-hallucinations-of-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2402.14683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14683"}},"official":{"repos":["wenhuang2000/vhtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-poison-large-language-models","slug":"learning-to-poison-large-language-models","title":"Learning to Poison Large Language Models for Downstream Manipulation","date":"2024-02-21","arxiv_id":"2402.13459","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-to-poison-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2402.13459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13459"}},"official":{"repos":["rookiezxy/gbtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieval-helps-or-hurts-a-deeper-dive-into","slug":"retrieval-helps-or-hurts-a-deeper-dive-into","title":"Retrieval Helps or Hurts? A Deeper Dive into the Efficacy of Retrieval Augmentation to Language Models","date":"2024-02-21","arxiv_id":"2402.13492","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/retrieval-helps-or-hurts-a-deeper-dive-into#ran","syntology_url":"https://syntology.ai/paper/2402.13492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13492"}},"official":{"repos":["megagonlabs/witqa"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/cognitive-visual-language-mapper-advancing","slug":"cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","arxiv_id":"2402.13561","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cognitive-visual-language-mapper-advancing#ran","syntology_url":"https://syntology.ai/paper/2402.13561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13561"}},"official":{"repos":["hitsz-tmg/cognitive-visual-language-mapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-building-multilingual-language-model","slug":"towards-building-multilingual-language-model","title":"Towards Building Multilingual Language Model for Medicine","date":"2024-02-21","arxiv_id":"2402.13963","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-building-multilingual-language-model#ran","syntology_url":"https://syntology.ai/paper/2402.13963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13963"}},"official":{"repos":["magic-ai4med/mmedlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fanoutqa-multi-hop-multi-document-question","slug":"fanoutqa-multi-hop-multi-document-question","title":"FanOutQA: A Multi-Hop, Multi-Document Question Answering Benchmark for Large Language Models","date":"2024-02-21","arxiv_id":"2402.14116","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fanoutqa-multi-hop-multi-document-question#ran","syntology_url":"https://syntology.ai/paper/2402.14116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14116"}},"official":{"repos":["zhudotexe/fanoutqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-finben-an-holistic-financial-benchmark","slug":"the-finben-an-holistic-financial-benchmark","title":"FinBen: A Holistic Financial Benchmark for Large Language Models","date":"2024-02-20","arxiv_id":"2402.12659","repositories_listed":2,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-finben-an-holistic-financial-benchmark#ran","syntology_url":"https://syntology.ai/paper/2402.12659","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12659"}},"official":{"repos":["the-finai/pixiu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/benchmarking-retrieval-augmented-generation","slug":"benchmarking-retrieval-augmented-generation","title":"Benchmarking Retrieval-Augmented Generation for Medicine","date":"2024-02-20","arxiv_id":"2402.13178","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2402.13178","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13178"}},"official":{"repos":["teddy-xionggz/medrag","teddy-xionggz/mirage"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mars-meaning-aware-response-scoring-for","slug":"mars-meaning-aware-response-scoring-for","title":"MARS: Meaning-Aware Response Scoring for Uncertainty Estimation in Generative LLMs","date":"2024-02-19","arxiv_id":"2402.11756","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mars-meaning-aware-response-scoring-for#ran","syntology_url":"https://syntology.ai/paper/2402.11756","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11756"}},"official":{"repos":["ybakman/llm_uncertainity"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/small-models-big-insights-leveraging-slim","slug":"small-models-big-insights-leveraging-slim","title":"Small Models, Big Insights: Leveraging Slim Proxy Models To Decide When and What to Retrieve for LLMs","date":"2024-02-19","arxiv_id":"2402.12052","repositories_listed":1,"syntology":{"n":15,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":15,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/small-models-big-insights-leveraging-slim#ran","syntology_url":"https://syntology.ai/paper/2402.12052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12052"}},"official":{"repos":["plageon/slimplm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/artifacts-or-abduction-how-do-llms-answer","slug":"artifacts-or-abduction-how-do-llms-answer","title":"Artifacts or Abduction: How Do LLMs Answer Multiple-Choice Questions Without the Question?","date":"2024-02-19","arxiv_id":"2402.12483","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/artifacts-or-abduction-how-do-llms-answer#ran","syntology_url":"https://syntology.ai/paper/2402.12483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12483"}},"official":{"repos":["nbalepur/mcqa-artifacts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/trustscore-reference-free-evaluation-of-llm","slug":"trustscore-reference-free-evaluation-of-llm","title":"TrustScore: Reference-Free Evaluation of LLM Response Trustworthiness","date":"2024-02-19","arxiv_id":"2402.12545","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trustscore-reference-free-evaluation-of-llm#ran","syntology_url":"https://syntology.ai/paper/2402.12545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12545"}},"official":{"repos":["dannalily/trustscore"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-knowledge-boundary-for-large","slug":"benchmarking-knowledge-boundary-for-large","title":"Benchmarking Knowledge Boundary for Large Language Models: A Different Perspective on Model Evaluation","date":"2024-02-18","arxiv_id":"2402.11493","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-knowledge-boundary-for-large#ran","syntology_url":"https://syntology.ai/paper/2402.11493","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11493"}},"official":{"repos":["pkulcwmzx/knowledge-boundary"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-from-failure-integrating-negative","slug":"learning-from-failure-integrating-negative","title":"Learning From Failure: Integrating Negative Examples when Fine-tuning Large Language Models as Agents","date":"2024-02-18","arxiv_id":"2402.11651","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-failure-integrating-negative#ran","syntology_url":"https://syntology.ai/paper/2402.11651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11651"}},"official":{"repos":["reason-wang/nat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/besa-pruning-large-language-models-with","slug":"besa-pruning-large-language-models-with","title":"BESA: Pruning Large Language Models with Blockwise Parameter-Efficient Sparsity Allocation","date":"2024-02-18","arxiv_id":"2402.16880","repositories_listed":2,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/besa-pruning-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2402.16880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16880"}},"official":{"repos":["linkanonymous/besa","opengvlab/llmprune-besa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/direct-evaluation-of-chain-of-thought-in","slug":"direct-evaluation-of-chain-of-thought-in","title":"Direct Evaluation of Chain-of-Thought in Multi-hop Reasoning with Knowledge Graphs","date":"2024-02-17","arxiv_id":"2402.11199","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/direct-evaluation-of-chain-of-thought-in#ran","syntology_url":"https://syntology.ai/paper/2402.11199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11199"}},"official":{"repos":["minhvuong2000/llmreasoncert"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-preference-alignment-remedies","slug":"multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","arxiv_id":"2402.10884","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-modal-preference-alignment-remedies#ran","syntology_url":"https://syntology.ai/paper/2402.10884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10884"}},"official":{"repos":["findalexli/mllm-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vqattack-transferable-adversarial-attacks-on","slug":"vqattack-transferable-adversarial-attacks-on","title":"VQAttack: Transferable Adversarial Attacks on Visual Question Answering via Pre-trained Models","date":"2024-02-16","arxiv_id":"2402.11083","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vqattack-transferable-adversarial-attacks-on#ran","syntology_url":"https://syntology.ai/paper/2402.11083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11083"}},"official":null}},{"url":"/paper/answer-is-all-you-need-instruction-following","slug":"answer-is-all-you-need-instruction-following","title":"Answer is All You Need: Instruction-following Text Embedding via Answering the Question","date":"2024-02-15","arxiv_id":"2402.09642","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/answer-is-all-you-need-instruction-following#ran","syntology_url":"https://syntology.ai/paper/2402.09642","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09642"}},"official":{"repos":["zhang-yu-wei/inbedder"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/controllm-crafting-diverse-personalities-for","slug":"controllm-crafting-diverse-personalities-for","title":"ControlLM: Crafting Diverse Personalities for Language Models","date":"2024-02-15","arxiv_id":"2402.10151","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/controllm-crafting-diverse-personalities-for#ran","syntology_url":"https://syntology.ai/paper/2402.10151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10151"}},"official":{"repos":["wengsyx/controllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/biomistral-a-collection-of-open-source","slug":"biomistral-a-collection-of-open-source","title":"BioMistral: A Collection of Open-Source Pretrained Large Language Models for Medical Domains","date":"2024-02-15","arxiv_id":"2402.10373","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/biomistral-a-collection-of-open-source#ran","syntology_url":"https://syntology.ai/paper/2402.10373","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10373"}},"official":{"repos":["biomistral/biomistral"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/omnimedvqa-a-new-large-scale-comprehensive","slug":"omnimedvqa-a-new-large-scale-comprehensive","title":"OmniMedVQA: A New Large-Scale Comprehensive Evaluation Benchmark for Medical LVLM","date":"2024-02-14","arxiv_id":"2402.09181","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omnimedvqa-a-new-large-scale-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.09181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09181"}},"official":{"repos":["opengvlab/multi-modality-arena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-faithful-and-robust-llm-specialists","slug":"towards-faithful-and-robust-llm-specialists","title":"Towards Faithful and Robust LLM Specialists for Evidence-Based Question-Answering","date":"2024-02-13","arxiv_id":"2402.08277","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-faithful-and-robust-llm-specialists#ran","syntology_url":"https://syntology.ai/paper/2402.08277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08277"}},"official":{"repos":["EdisonNi-hku/Robust_Evidence_Based_QA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-perceptual-limitation-of-multimodal","slug":"exploring-perceptual-limitation-of-multimodal","title":"Exploring Perceptual Limitation of Multimodal Large Language Models","date":"2024-02-12","arxiv_id":"2402.07384","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/exploring-perceptual-limitation-of-multimodal#ran","syntology_url":"https://syntology.ai/paper/2402.07384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07384"}},"official":{"repos":["saccharomycetes/mllm-perceptual-limitation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/anchor-based-large-language-models","slug":"anchor-based-large-language-models","title":"Anchor-based Large Language Models","date":"2024-02-12","arxiv_id":"2402.07616","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/anchor-based-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2402.07616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07616"}},"official":{"repos":["pangjh3/anllm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/g-retriever-retrieval-augmented-generation","slug":"g-retriever-retrieval-augmented-generation","title":"G-Retriever: Retrieval-Augmented Generation for Textual Graph Understanding and Question Answering","date":"2024-02-12","arxiv_id":"2402.07630","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/g-retriever-retrieval-augmented-generation#ran","syntology_url":"https://syntology.ai/paper/2402.07630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07630"}},"official":{"repos":["xiaoxinhe/g-retriever"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prismatic-vlms-investigating-the-design-space","slug":"prismatic-vlms-investigating-the-design-space","title":"Prismatic VLMs: Investigating the Design Space of Visually-Conditioned Language Models","date":"2024-02-12","arxiv_id":"2402.07865","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prismatic-vlms-investigating-the-design-space#ran","syntology_url":"https://syntology.ai/paper/2402.07865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07865"}},"official":{"repos":["tri-ml/prismatic-vlms","tri-ml/vlm-evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-benchmark-for-multi-modal-foundation-models","slug":"a-benchmark-for-multi-modal-foundation-models","title":"Q-Bench+: A Benchmark for Multi-modal Foundation Models on Low-level Vision from Single Images to Pairs","date":"2024-02-11","arxiv_id":"2402.07116","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-benchmark-for-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2402.07116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07116"}},"official":{"repos":["Q-Future/Q-Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/graphtranslator-aligning-graph-model-to-large","slug":"graphtranslator-aligning-graph-model-to-large","title":"GraphTranslator: Aligning Graph Model to Large Language Model for Open-ended Tasks","date":"2024-02-11","arxiv_id":"2402.07197","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graphtranslator-aligning-graph-model-to-large#ran","syntology_url":"https://syntology.ai/paper/2402.07197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07197"}},"official":{"repos":["alibaba/graphtranslator"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crema-multimodal-compositional-video","slug":"crema-multimodal-compositional-video","title":"CREMA: Generalizable and Efficient Video-Language Reasoning via Multimodal Modular Fusion","date":"2024-02-08","arxiv_id":"2402.05889","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crema-multimodal-compositional-video#ran","syntology_url":"https://syntology.ai/paper/2402.05889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05889"}},"official":{"repos":["Yui010206/CREMA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/training-language-models-to-generate-text","slug":"training-language-models-to-generate-text","title":"Training Language Models to Generate Text with Citations via Fine-grained Rewards","date":"2024-02-06","arxiv_id":"2402.04315","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/training-language-models-to-generate-text#ran","syntology_url":"https://syntology.ai/paper/2402.04315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04315"}},"official":{"repos":["hcy123902/atg-w-fg-rw"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gerea-question-aware-prompt-captions-for#ran","syntology_url":"https://syntology.ai/paper/2402.02503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02503"}},"official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-complex-question-answering-over","slug":"enhancing-complex-question-answering-over","title":"Enhancing Complex Question Answering over Knowledge Graphs through Evidence Pattern Retrieval","date":"2024-02-03","arxiv_id":"2402.02175","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-complex-question-answering-over#ran","syntology_url":"https://syntology.ai/paper/2402.02175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02175"}},"official":{"repos":["nju-websoft/epr-kgqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cabinet-content-relevance-based-noise","slug":"cabinet-content-relevance-based-noise","title":"CABINET: Content Relevance based Noise Reduction for Table Question Answering","date":"2024-02-02","arxiv_id":"2402.01155","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cabinet-content-relevance-based-noise#ran","syntology_url":"https://syntology.ai/paper/2402.01155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01155"}},"official":{"repos":["sohanpatnaik106/cabinet_qa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/raptor-recursive-abstractive-processing-for","slug":"raptor-recursive-abstractive-processing-for","title":"RAPTOR: Recursive Abstractive Processing for Tree-Organized Retrieval","date":"2024-01-31","arxiv_id":"2401.18059","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/raptor-recursive-abstractive-processing-for#ran","syntology_url":"https://syntology.ai/paper/2401.18059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.18059"}},"official":{"repos":["parthsarthi03/RAPTOR"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crud-rag-a-comprehensive-chinese-benchmark","slug":"crud-rag-a-comprehensive-chinese-benchmark","title":"CRUD-RAG: A Comprehensive Chinese Benchmark for Retrieval-Augmented Generation of Large Language Models","date":"2024-01-30","arxiv_id":"2401.17043","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/crud-rag-a-comprehensive-chinese-benchmark#ran","syntology_url":"https://syntology.ai/paper/2401.17043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17043"}},"official":{"repos":["iaar-shanghai/crud_rag"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-consistent-natural-language","slug":"towards-consistent-natural-language","title":"Towards Consistent Natural-Language Explanations via Explanation-Consistency Finetuning","date":"2024-01-25","arxiv_id":"2401.13986","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-consistent-natural-language#ran","syntology_url":"https://syntology.ai/paper/2401.13986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13986"}},"official":{"repos":["yandachen/explanation-consistency-finetuning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/longhealth-a-question-answering-benchmark","slug":"longhealth-a-question-answering-benchmark","title":"LongHealth: A Question Answering Benchmark with Long Clinical Documents","date":"2024-01-25","arxiv_id":"2401.14490","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/longhealth-a-question-answering-benchmark#ran","syntology_url":"https://syntology.ai/paper/2401.14490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14490"}},"official":{"repos":["kbressem/longhealth"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/seer-facilitating-structured-reasoning-and","slug":"seer-facilitating-structured-reasoning-and","title":"SEER: Facilitating Structured Reasoning and Explanation via Reinforcement Learning","date":"2024-01-24","arxiv_id":"2401.13246","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":8,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":2,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/seer-facilitating-structured-reasoning-and#ran","syntology_url":"https://syntology.ai/paper/2401.13246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13246"}},"official":{"repos":["chen-gx/seer"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/can-ai-assistants-know-what-they-don-t-know","slug":"can-ai-assistants-know-what-they-don-t-know","title":"Can AI Assistants Know What They Don't Know?","date":"2024-01-24","arxiv_id":"2401.13275","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-ai-assistants-know-what-they-don-t-know#ran","syntology_url":"https://syntology.ai/paper/2401.13275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13275"}},"official":{"repos":["openmoss/say-i-dont-know"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/trove-inducing-verifiable-and-efficient","slug":"trove-inducing-verifiable-and-efficient","title":"TroVE: Inducing Verifiable and Efficient Toolboxes for Solving Programmatic Tasks","date":"2024-01-23","arxiv_id":"2401.12869","repositories_listed":1,"syntology":{"n":19,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trove-inducing-verifiable-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2401.12869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12869"}},"official":{"repos":["zorazrw/trove"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-fusion-of-large-language-models","slug":"knowledge-fusion-of-large-language-models","title":"Knowledge Fusion of Large Language Models","date":"2024-01-19","arxiv_id":"2401.10491","repositories_listed":3,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-fusion-of-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.10491","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.10491"}},"official":{"repos":["fanqiwan/fusellm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/question-answer-cross-language-image-matching","slug":"question-answer-cross-language-image-matching","title":"Question-Answer Cross Language Image Matching for Weakly Supervised Semantic Segmentation","date":"2024-01-18","arxiv_id":"2401.09883","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/question-answer-cross-language-image-matching#ran","syntology_url":"https://syntology.ai/paper/2401.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09883"}},"official":{"repos":["cvi-szu/qa-clims"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tuning-language-models-by-proxy","slug":"tuning-language-models-by-proxy","title":"Tuning Language Models by Proxy","date":"2024-01-16","arxiv_id":"2401.08565","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tuning-language-models-by-proxy#ran","syntology_url":"https://syntology.ai/paper/2401.08565","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.08565"}},"official":{"repos":["alisawuffles/proxy-tuning"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-large-language-models-explain-themselves-1","slug":"can-large-language-models-explain-themselves-1","title":"Are self-explanations from Large Language Models faithful?","date":"2024-01-15","arxiv_id":"2401.07927","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-large-language-models-explain-themselves-1#ran","syntology_url":"https://syntology.ai/paper/2401.07927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07927"}},"official":{"repos":["AndreasMadsen/llm-introspection"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-external-knowledge-resources-to","slug":"leveraging-external-knowledge-resources-to","title":"Towards Efficient Methods in Medical Question Answering using Knowledge Graph Embeddings","date":"2024-01-15","arxiv_id":"2401.07977","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leveraging-external-knowledge-resources-to#ran","syntology_url":"https://syntology.ai/paper/2401.07977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07977"}},"official":{"repos":["saptarshi059/cdqa-project"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ehragent-code-empowers-large-language-models","slug":"ehragent-code-empowers-large-language-models","title":"EHRAgent: Code Empowers Large Language Models for Few-shot Complex Tabular Reasoning on Electronic Health Records","date":"2024-01-13","arxiv_id":"2401.07128","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ehragent-code-empowers-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.07128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07128"}},"official":{"repos":["wshi83/ehragent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-unreasonable-effectiveness-of-easy","slug":"the-unreasonable-effectiveness-of-easy","title":"The Unreasonable Effectiveness of Easy Training Data for Hard Tasks","date":"2024-01-12","arxiv_id":"2401.06751","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-unreasonable-effectiveness-of-easy#ran","syntology_url":"https://syntology.ai/paper/2401.06751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06751"}},"official":{"repos":["allenai/easy-to-hard-generalization"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-retrieval-for-knowledge-based","slug":"cross-modal-retrieval-for-knowledge-based","title":"Cross-modal Retrieval for Knowledge-based Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05736","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/cross-modal-retrieval-for-knowledge-based#ran","syntology_url":"https://syntology.ai/paper/2401.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05736"}},"official":{"repos":["paullerner/viquae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-benchmark-in-medical-visual","slug":"hallucination-benchmark-in-medical-visual","title":"Hallucination Benchmark in Medical Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05827","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallucination-benchmark-in-medical-visual#ran","syntology_url":"https://syntology.ai/paper/2401.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05827"}},"official":{"repos":["knowlab/halt-medvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-large-language-models-via-fine","slug":"improving-large-language-models-via-fine","title":"Improving Large Language Models via Fine-grained Reinforcement Learning with Minimum Editing Constraint","date":"2024-01-11","arxiv_id":"2401.06081","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-large-language-models-via-fine#ran","syntology_url":"https://syntology.ai/paper/2401.06081","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06081"}},"official":{"repos":["rucaibox/rlmec"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/divide-and-conquer-for-large-language-models","slug":"divide-and-conquer-for-large-language-models","title":"DCR: Divide-and-Conquer Reasoning for Multi-choice Question Answering with LLMs","date":"2024-01-10","arxiv_id":"2401.05190","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/divide-and-conquer-for-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2401.05190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05190"}},"official":{"repos":["aimijie/dcr","aimijie/divide-and-conquer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autoact-automatic-agent-learning-from-scratch","slug":"autoact-automatic-agent-learning-from-scratch","title":"AutoAct: Automatic Agent Learning from Scratch for QA via Self-Planning","date":"2024-01-10","arxiv_id":"2401.05268","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/autoact-automatic-agent-learning-from-scratch#ran","syntology_url":"https://syntology.ai/paper/2401.05268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05268"}},"official":{"repos":["zjunlp/autoact"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-table-evolving-tables-in-the","slug":"chain-of-table-evolving-tables-in-the","title":"Chain-of-Table: Evolving Tables in the Reasoning Chain for Table Understanding","date":"2024-01-09","arxiv_id":"2401.04398","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-table-evolving-tables-in-the#ran","syntology_url":"https://syntology.ai/paper/2401.04398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04398"}},"official":null}},{"url":"/paper/the-critique-of-critique","slug":"the-critique-of-critique","title":"The Critique of Critique","date":"2024-01-09","arxiv_id":"2401.04518","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-critique-of-critique#ran","syntology_url":"https://syntology.ai/paper/2401.04518","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04518"}},"official":{"repos":["gair-nlp/metacritique"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-editing-can-hurt-general-abilities-of","slug":"model-editing-can-hurt-general-abilities-of","title":"Model Editing Harms General Abilities of Large Language Models: Regularization to the Rescue","date":"2024-01-09","arxiv_id":"2401.04700","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-editing-can-hurt-general-abilities-of#ran","syntology_url":"https://syntology.ai/paper/2401.04700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04700"}},"official":{"repos":["jasonforjoy/model-editing-hurt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/mixtral-of-experts","slug":"mixtral-of-experts","title":"Mixtral of Experts","date":"2024-01-08","arxiv_id":"2401.04088","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/mixtral-of-experts#ran","syntology_url":"https://syntology.ai/paper/2401.04088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04088"}},"official":null}},{"url":"/paper/glance-and-focus-memory-prompting-for-multi-1","slug":"glance-and-focus-memory-prompting-for-multi-1","title":"Glance and Focus: Memory Prompting for Multi-Event Video Question Answering","date":"2024-01-03","arxiv_id":"2401.01529","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":2,"n_ran_checked":14,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":13,"n_pointer_only":7,"phrase":"14 ran (of which 2 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 1 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/glance-and-focus-memory-prompting-for-multi-1#ran","syntology_url":"https://syntology.ai/paper/2401.01529","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01529"}},"official":{"repos":["byz0e/glance-focus"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":2,"n_ran_no_instrument_failure":14,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if","slug":"gpt-4v-ision-is-a-generalist-web-agent-if","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","date":"2024-01-03","arxiv_id":"2401.01614","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if#ran","syntology_url":"https://syntology.ai/paper/2401.01614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01614"}},"official":{"repos":["osu-nlp-group/seeact"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-the-impact-of-false-negatives-in","slug":"mitigating-the-impact-of-false-negatives-in","title":"Mitigating the Impact of False Negatives in Dense Retrieval with Contrastive Confidence Regularization","date":"2023-12-30","arxiv_id":"2401.00165","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mitigating-the-impact-of-false-negatives-in#ran","syntology_url":"https://syntology.ai/paper/2401.00165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.00165"}},"official":{"repos":["wangskygit/passage-sieve"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/tinygpt-v-efficient-multimodal-large-language","slug":"tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","arxiv_id":"2312.16862","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tinygpt-v-efficient-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.16862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16862"}},"official":{"repos":["dlyuangod/tinygpt-v"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-llm-framework-for-long-range-video","slug":"a-simple-llm-framework-for-long-range-video","title":"A Simple LLM Framework for Long-Range Video Question-Answering","date":"2023-12-28","arxiv_id":"2312.17235","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-simple-llm-framework-for-long-range-video#ran","syntology_url":"https://syntology.ai/paper/2312.17235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17235"}},"official":{"repos":["ceezh/llovi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/secqa-a-concise-question-answering-dataset","slug":"secqa-a-concise-question-answering-dataset","title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","date":"2023-12-26","arxiv_id":"2312.15838","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/secqa-a-concise-question-answering-dataset#ran","syntology_url":"https://syntology.ai/paper/2312.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15838"}},"official":{"repos":["zefang-liu/lm-evaluation-harness"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/supervised-knowledge-makes-large-language","slug":"supervised-knowledge-makes-large-language","title":"Supervised Knowledge Makes Large Language Models Better In-context Learners","date":"2023-12-26","arxiv_id":"2312.15918","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/supervised-knowledge-makes-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.15918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15918"}},"official":{"repos":["yanglinyi/supervised-knowledge-makes-large-language-models-better-in-context-learners"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pokemqa-programmable-knowledge-editing-for","slug":"pokemqa-programmable-knowledge-editing-for","title":"PokeMQA: Programmable knowledge editing for Multi-hop Question Answering","date":"2023-12-23","arxiv_id":"2312.15194","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pokemqa-programmable-knowledge-editing-for#ran","syntology_url":"https://syntology.ai/paper/2312.15194","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15194"}},"official":{"repos":["hengrui-gu/pokemqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lingoqa-video-question-answering-for","slug":"lingoqa-video-question-answering-for","title":"LingoQA: Visual Question Answering for Autonomous Driving","date":"2023-12-21","arxiv_id":"2312.14115","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lingoqa-video-question-answering-for#ran","syntology_url":"https://syntology.ai/paper/2312.14115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14115"}},"official":{"repos":["wayveai/lingoqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vcoder-versatile-vision-encoders-for","slug":"vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","arxiv_id":"2312.14233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vcoder-versatile-vision-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2312.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14233"}},"official":{"repos":["shi-labs/vcoder"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/lookahead-an-inference-acceleration-framework","slug":"lookahead-an-inference-acceleration-framework","title":"Lookahead: An Inference Acceleration Framework for Large Language Model with Lossless Generation Accuracy","date":"2023-12-20","arxiv_id":"2312.12728","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lookahead-an-inference-acceleration-framework#ran","syntology_url":"https://syntology.ai/paper/2312.12728","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12728"}},"official":{"repos":["alipay/PainlessInferenceAcceleration"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-multimodal-models-are-in-context","slug":"generative-multimodal-models-are-in-context","title":"Generative Multimodal Models are In-Context Learners","date":"2023-12-20","arxiv_id":"2312.13286","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generative-multimodal-models-are-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.13286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13286"}},"official":{"repos":["baaivision/emu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/melo-enhancing-model-editing-with-neuron","slug":"melo-enhancing-model-editing-with-neuron","title":"MELO: Enhancing Model Editing with Neuron-Indexed Dynamic LoRA","date":"2023-12-19","arxiv_id":"2312.11795","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/melo-enhancing-model-editing-with-neuron#ran","syntology_url":"https://syntology.ai/paper/2312.11795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.11795"}},"official":{"repos":["bruthyu/melo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/earthvqa-towards-queryable-earth-via","slug":"earthvqa-towards-queryable-earth-via","title":"EarthVQA: Towards Queryable Earth via Relational Reasoning-Based Remote Sensing Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earthvqa-towards-queryable-earth-via#ran","syntology_url":"https://syntology.ai/paper/2312.12222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12222"}},"official":{"repos":["Junjue-Wang/EarthVQA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-early-detection-of-hallucinations-in","slug":"on-early-detection-of-hallucinations-in","title":"On Early Detection of Hallucinations in Factual Question Answering","date":"2023-12-19","arxiv_id":"2312.14183","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-early-detection-of-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2312.14183","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14183"}},"official":{"repos":["amazon-science/llm-hallucinations-factual-qa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/icd-lm-configuring-vision-language-in-context","slug":"icd-lm-configuring-vision-language-in-context","title":"Lever LM: Configuring In-Context Sequence to Lever Large Vision Language Models","date":"2023-12-15","arxiv_id":"2312.10104","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/icd-lm-configuring-vision-language-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.10104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10104"}},"official":{"repos":["forjadeforest/icd-lm","forjadeforest/lever-lm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"894fb3c3360357e4b84187537be650c4e3b99f1a0eae4754e652278e2bcc1a4f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}