{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/ran/3","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":3,"rows_per_page":100,"rows":[201,276],"of":276,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination/papers/ran/1","prev":"/task/hallucination/papers/ran/2","next":null,"papers":[{"url":"/paper/finding-and-editing-multi-modal-neurons-in","slug":"finding-and-editing-multi-modal-neurons-in","title":"Finding and Editing Multi-Modal Neurons in Pre-Trained Transformers","date":"2023-11-13","arxiv_id":"2311.07470","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/finding-and-editing-multi-modal-neurons-in#ran","syntology_url":"https://syntology.ai/paper/2311.07470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07470"}},"official":{"repos":["opanhw/MM_Neurons"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-large-language-model-for","slug":"collaborative-large-language-model-for","title":"Collaborative Large Language Model for Recommender Systems","date":"2023-11-02","arxiv_id":"2311.01343","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborative-large-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2311.01343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01343"}},"official":{"repos":["yaochenzhu/llm4rec"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/distil-whisper-robust-knowledge-distillation","slug":"distil-whisper-robust-knowledge-distillation","title":"Distil-Whisper: Robust Knowledge Distillation via Large-Scale Pseudo Labelling","date":"2023-11-01","arxiv_id":"2311.00430","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/distil-whisper-robust-knowledge-distillation#ran","syntology_url":"https://syntology.ai/paper/2311.00430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00430"}},"official":{"repos":["huggingface/distil-whisper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/synthetic-imitation-edit-feedback-for-factual","slug":"synthetic-imitation-edit-feedback-for-factual","title":"Synthetic Imitation Edit Feedback for Factual Alignment in Clinical Summarization","date":"2023-10-30","arxiv_id":"2310.20033","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":3,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/synthetic-imitation-edit-feedback-for-factual#ran","syntology_url":"https://syntology.ai/paper/2310.20033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20033"}},"official":{"repos":["seasonyao/learnfromhumanedit"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lightlm-a-lightweight-deep-and-narrow","slug":"lightlm-a-lightweight-deep-and-narrow","title":"LightLM: A Lightweight Deep and Narrow Language Model for Generative Recommendation","date":"2023-10-26","arxiv_id":"2310.17488","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lightlm-a-lightweight-deep-and-narrow#ran","syntology_url":"https://syntology.ai/paper/2310.17488","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17488"}},"official":{"repos":["dongyuanjushi/lightlm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/woodpecker-hallucination-correction-for","slug":"woodpecker-hallucination-correction-for","title":"Woodpecker: Hallucination Correction for Multimodal Large Language Models","date":"2023-10-24","arxiv_id":"2310.16045","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/woodpecker-hallucination-correction-for#ran","syntology_url":"https://syntology.ai/paper/2310.16045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16045"}},"official":{"repos":["bradyfu/woodpecker"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-hallucinate-but-may-excel-at","slug":"language-models-hallucinate-but-may-excel-at","title":"Language Models Hallucinate, but May Excel at Fact Verification","date":"2023-10-23","arxiv_id":"2310.14564","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/language-models-hallucinate-but-may-excel-at#ran","syntology_url":"https://syntology.ai/paper/2310.14564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14564"}},"official":{"repos":["jianguanthu/llmforfv"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hallusionbench-you-see-what-you-think-or-you","slug":"hallusionbench-you-see-what-you-think-or-you","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","date":"2023-10-23","arxiv_id":"2310.14566","repositories_listed":9,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/hallusionbench-you-see-what-you-think-or-you#ran","syntology_url":"https://syntology.ai/paper/2310.14566","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.14566"}},"official":{"repos":["tianyi-lab/hallusionbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/maf-multi-aspect-feedback-for-improving","slug":"maf-multi-aspect-feedback-for-improving","title":"MAF: Multi-Aspect Feedback for Improving Reasoning in Large Language Models","date":"2023-10-19","arxiv_id":"2310.12426","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/maf-multi-aspect-feedback-for-improving#ran","syntology_url":"https://syntology.ai/paper/2310.12426","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12426"}},"official":{"repos":["deepakn97/maf"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unveiling-the-siren-s-song-towards-reliable","slug":"unveiling-the-siren-s-song-towards-reliable","title":"FactCHD: Benchmarking Fact-Conflicting Hallucination Detection","date":"2023-10-18","arxiv_id":"2310.12086","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unveiling-the-siren-s-song-towards-reliable#ran","syntology_url":"https://syntology.ai/paper/2310.12086","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12086"}},"official":{"repos":["zjunlp/factchd"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lidar-based-4d-occupancy-completion-and","slug":"lidar-based-4d-occupancy-completion-and","title":"LiDAR-based 4D Occupancy Completion and Forecasting","date":"2023-10-17","arxiv_id":"2310.11239","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lidar-based-4d-occupancy-completion-and#ran","syntology_url":"https://syntology.ai/paper/2310.11239","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11239"}},"official":{"repos":["ai4ce/occ4cast"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/regavae-a-retrieval-augmented-gaussian","slug":"regavae-a-retrieval-augmented-gaussian","title":"RegaVAE: A Retrieval-Augmented Gaussian Mixture Variational Auto-Encoder for Language Modeling","date":"2023-10-16","arxiv_id":"2310.10567","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regavae-a-retrieval-augmented-gaussian#ran","syntology_url":"https://syntology.ai/paper/2310.10567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10567"}},"official":{"repos":["trustedllm/regavae"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/kcts-knowledge-constrained-tree-search","slug":"kcts-knowledge-constrained-tree-search","title":"KCTS: Knowledge-Constrained Tree Search Decoding with Token-Level Hallucination Detection","date":"2023-10-13","arxiv_id":"2310.09044","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kcts-knowledge-constrained-tree-search#ran","syntology_url":"https://syntology.ai/paper/2310.09044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09044"}},"official":{"repos":["hkust-knowcomp/knowledge-constrained-decoding"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/kelly-is-a-warm-person-joseph-is-a-role-model","slug":"kelly-is-a-warm-person-joseph-is-a-role-model","title":"\"Kelly is a Warm Person, Joseph is a Role Model\": Gender Biases in LLM-Generated Reference Letters","date":"2023-10-13","arxiv_id":"2310.09219","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/kelly-is-a-warm-person-joseph-is-a-role-model#ran","syntology_url":"https://syntology.ai/paper/2310.09219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09219"}},"official":{"repos":["uclanlp/biases-llm-reference-letters"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ferret-refer-and-ground-anything-anywhere-at","slug":"ferret-refer-and-ground-anything-anywhere-at","title":"Ferret: Refer and Ground Anything Anywhere at Any Granularity","date":"2023-10-11","arxiv_id":"2310.07704","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ferret-refer-and-ground-anything-anywhere-at#ran","syntology_url":"https://syntology.ai/paper/2310.07704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07704"}},"official":{"repos":["apple/ml-ferret"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-new-benchmark-and-reverse-validation-method","slug":"a-new-benchmark-and-reverse-validation-method","title":"A New Benchmark and Reverse Validation Method for Passage-level Hallucination Detection","date":"2023-10-10","arxiv_id":"2310.06498","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-new-benchmark-and-reverse-validation-method#ran","syntology_url":"https://syntology.ai/paper/2310.06498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.06498"}},"official":{"repos":["maybenotime/phd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-natural-language-inference-for","slug":"chain-of-natural-language-inference-for","title":"Chain of Natural Language Inference for Reducing Large Language Model Ungrounded Hallucinations","date":"2023-10-06","arxiv_id":"2310.03951","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chain-of-natural-language-inference-for#ran","syntology_url":"https://syntology.ai/paper/2310.03951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03951"}},"official":{"repos":["microsoft/conli_hallucination"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-as-ai","slug":"benchmarking-large-language-models-as-ai","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","date":"2023-10-05","arxiv_id":"2310.03302","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-large-language-models-as-ai#ran","syntology_url":"https://syntology.ai/paper/2310.03302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03302"}},"official":{"repos":["snap-stanford/mlagentbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-hallucinations-in-chinese-large","slug":"evaluating-hallucinations-in-chinese-large","title":"Evaluating Hallucinations in Chinese Large Language Models","date":"2023-10-05","arxiv_id":"2310.03368","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-hallucinations-in-chinese-large#ran","syntology_url":"https://syntology.ai/paper/2310.03368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03368"}},"official":{"repos":["xiami2019/halluqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/halle-switch-rethinking-and-controlling","slug":"halle-switch-rethinking-and-controlling","title":"HallE-Control: Controlling Object Hallucination in Large Multimodal Models","date":"2023-10-03","arxiv_id":"2310.01779","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/halle-switch-rethinking-and-controlling#ran","syntology_url":"https://syntology.ai/paper/2310.01779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01779"}},"official":{"repos":["bronyayang/HallE_Switch","bronyayang/halle_control"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/btr-binary-token-representations-for","slug":"btr-binary-token-representations-for","title":"BTR: Binary Token Representations for Efficient Retrieval Augmented Language Models","date":"2023-10-02","arxiv_id":"2310.01329","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/btr-binary-token-representations-for#ran","syntology_url":"https://syntology.ai/paper/2310.01329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01329"}},"official":{"repos":["csarron/btr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-lies-hallucinations-are-not-bugs-but","slug":"llm-lies-hallucinations-are-not-bugs-but","title":"LLM Lies: Hallucinations are not Bugs, but Features as Adversarial Examples","date":"2023-10-02","arxiv_id":"2310.01469","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-lies-hallucinations-are-not-bugs-but#ran","syntology_url":"https://syntology.ai/paper/2310.01469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01469"}},"official":{"repos":["pku-yuangroup/hallucination-attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-and-mitigating-object-hallucination","slug":"analyzing-and-mitigating-object-hallucination","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","date":"2023-10-01","arxiv_id":"2310.00754","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/analyzing-and-mitigating-object-hallucination#ran","syntology_url":"https://syntology.ai/paper/2310.00754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00754"}},"official":{"repos":["yiyangzhou/lure"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/see-beyond-seeing-robust-3d-object-detection","slug":"see-beyond-seeing-robust-3d-object-detection","title":"Robust 3D Object Detection from LiDAR-Radar Point Clouds via Cross-Modal Feature Augmentation","date":"2023-09-29","arxiv_id":"2309.17336","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/see-beyond-seeing-robust-3d-object-detection#ran","syntology_url":"https://syntology.ai/paper/2309.17336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17336"}},"official":{"repos":["djning/see_beyond_seeing"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-cross-view-representation-1","slug":"self-supervised-cross-view-representation-1","title":"Self-supervised Cross-view Representation Reconstruction for Change Captioning","date":"2023-09-28","arxiv_id":"2309.16283","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/self-supervised-cross-view-representation-1#ran","syntology_url":"https://syntology.ai/paper/2309.16283","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16283"}},"official":{"repos":["tuyunbin/scorer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lyra-orchestrating-dual-correction-in","slug":"lyra-orchestrating-dual-correction-in","title":"Lyra: Orchestrating Dual Correction in Automated Theorem Proving","date":"2023-09-27","arxiv_id":"2309.15806","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lyra-orchestrating-dual-correction-in#ran","syntology_url":"https://syntology.ai/paper/2309.15806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15806"}},"official":{"repos":["chuanyang-zheng/lyra-theorem-prover"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bamboo-a-comprehensive-benchmark-for","slug":"bamboo-a-comprehensive-benchmark-for","title":"BAMBOO: A Comprehensive Benchmark for Evaluating Long Text Modeling Capacities of Large Language Models","date":"2023-09-23","arxiv_id":"2309.13345","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bamboo-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2309.13345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13345"}},"official":{"repos":["rucaibox/bamboo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmicl-empowering-vision-language-model-with","slug":"mmicl-empowering-vision-language-model-with","title":"MMICL: Empowering Vision-language Model with Multi-Modal In-Context Learning","date":"2023-09-14","arxiv_id":"2309.07915","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mmicl-empowering-vision-language-model-with#ran","syntology_url":"https://syntology.ai/paper/2309.07915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.07915"}},"official":{"repos":["haozhezhao/mic","pkunlp-icler/mic"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tegit-generating-high-quality-instruction","slug":"tegit-generating-high-quality-instruction","title":"DoG-Instruct: Towards Premium Instruction-Tuning Data via Text-Grounded Instruction Wrapping","date":"2023-09-11","arxiv_id":"2309.05447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tegit-generating-high-quality-instruction#ran","syntology_url":"https://syntology.ai/paper/2309.05447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05447"}},"official":{"repos":["bahuia/dog-instruct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-tuning-large-language-models-with","slug":"knowledge-tuning-large-language-models-with","title":"Knowledge-tuning Large Language Models with Structured Medical Knowledge Bases for Reliable Response Generation in Chinese","date":"2023-09-08","arxiv_id":"2309.04175","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-tuning-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2309.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04175"}},"official":null}},{"url":"/paper/benchmarking-large-language-models-in","slug":"benchmarking-large-language-models-in","title":"Benchmarking Large Language Models in Retrieval-Augmented Generation","date":"2023-09-04","arxiv_id":"2309.01431","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2309.01431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01431"}},"official":{"repos":["chen700564/RGB"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vigc-visual-instruction-generation-and","slug":"vigc-visual-instruction-generation-and","title":"VIGC: Visual Instruction Generation and Correction","date":"2023-08-24","arxiv_id":"2308.12714","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vigc-visual-instruction-generation-and#ran","syntology_url":"https://syntology.ai/paper/2308.12714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12714"}},"official":{"repos":["opendatalab/vigc"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/prefer-prompt-ensemble-learning-via-feedback","slug":"prefer-prompt-ensemble-learning-via-feedback","title":"PREFER: Prompt Ensemble Learning via Feedback-Reflect-Refine","date":"2023-08-23","arxiv_id":"2308.12033","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prefer-prompt-ensemble-learning-via-feedback#ran","syntology_url":"https://syntology.ai/paper/2308.12033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12033"}},"official":{"repos":["zcrwind/prefer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lan-hdr-luminance-based-alignment-network-for","slug":"lan-hdr-luminance-based-alignment-network-for","title":"LAN-HDR: Luminance-based Alignment Network for High Dynamic Range Video Reconstruction","date":"2023-08-22","arxiv_id":"2308.11116","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":1,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":18,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/lan-hdr-luminance-based-alignment-network-for#ran","syntology_url":"https://syntology.ai/paper/2308.11116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11116"}},"official":{"repos":["haesoochung/lan-hdr"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":1,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/mindmap-knowledge-graph-prompting-sparks","slug":"mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","arxiv_id":"2308.09729","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mindmap-knowledge-graph-prompting-sparks#ran","syntology_url":"https://syntology.ai/paper/2308.09729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09729"}},"official":{"repos":["wyl-willing/MindMap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/transferable-decoding-with-visual-entities","slug":"transferable-decoding-with-visual-entities","title":"Transferable Decoding with Visual Entities for Zero-Shot Image Captioning","date":"2023-07-31","arxiv_id":"2307.16525","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transferable-decoding-with-visual-entities#ran","syntology_url":"https://syntology.ai/paper/2307.16525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16525"}},"official":{"repos":["feielysia/viecap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-llms-possess-a-personality-making-the-mbti","slug":"do-llms-possess-a-personality-making-the-mbti","title":"Do LLMs Possess a Personality? Making the MBTI Test an Amazing Evaluation for Large Language Models","date":"2023-07-30","arxiv_id":"2307.16180","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/do-llms-possess-a-personality-making-the-mbti#ran","syntology_url":"https://syntology.ai/paper/2307.16180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16180"}},"official":{"repos":["harderthenharder/transformers_tasks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pac-neural-prediction-set-learning-to","slug":"pac-neural-prediction-set-learning-to","title":"Selective Generation for Controllable Language Models","date":"2023-07-18","arxiv_id":"2307.09254","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pac-neural-prediction-set-learning-to#ran","syntology_url":"https://syntology.ai/paper/2307.09254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.09254"}},"official":{"repos":["ml-postech/selective-generation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/think-on-graph-deep-and-responsible-reasoning","slug":"think-on-graph-deep-and-responsible-reasoning","title":"Think-on-Graph: Deep and Responsible Reasoning of Large Language Model on Knowledge Graph","date":"2023-07-15","arxiv_id":"2307.07697","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/think-on-graph-deep-and-responsible-reasoning#ran","syntology_url":"https://syntology.ai/paper/2307.07697","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.07697"}},"official":{"repos":["gasolsun36/tog","idea-finai/tog"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["community","official"]}}},{"url":"/paper/prompts-should-not-be-seen-as-secrets","slug":"prompts-should-not-be-seen-as-secrets","title":"Effective Prompt Extraction from Language Models","date":"2023-07-13","arxiv_id":"2307.06865","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompts-should-not-be-seen-as-secrets#ran","syntology_url":"https://syntology.ai/paper/2307.06865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.06865"}},"official":{"repos":["y0mingzhang/prompt-extraction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toolqa-a-dataset-for-llm-question-answering-1","slug":"toolqa-a-dataset-for-llm-question-answering-1","title":"ToolQA: A Dataset for LLM Question Answering with External Tools","date":"2023-06-23","arxiv_id":"2306.13304","repositories_listed":2,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":13,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/toolqa-a-dataset-for-llm-question-answering-1#ran","syntology_url":"https://syntology.ai/paper/2306.13304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.13304"}},"official":{"repos":["night-chen/toolqa"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/do-language-models-know-when-they-re","slug":"do-language-models-know-when-they-re","title":"Do Language Models Know When They're Hallucinating References?","date":"2023-05-29","arxiv_id":"2305.18248","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/do-language-models-know-when-they-re#ran","syntology_url":"https://syntology.ai/paper/2305.18248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18248"}},"official":{"repos":["microsoft/hallucinated-references"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaplanner-adaptive-planning-from-feedback-1","slug":"adaplanner-adaptive-planning-from-feedback-1","title":"AdaPlanner: Adaptive Planning from Feedback with Language Models","date":"2023-05-26","arxiv_id":"2305.16653","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaplanner-adaptive-planning-from-feedback-1#ran","syntology_url":"https://syntology.ai/paper/2305.16653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16653"}},"official":{"repos":["haotiansun14/adaplanner"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enabling-large-language-models-to-generate","slug":"enabling-large-language-models-to-generate","title":"Enabling Large Language Models to Generate Text with Citations","date":"2023-05-24","arxiv_id":"2305.14627","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enabling-large-language-models-to-generate#ran","syntology_url":"https://syntology.ai/paper/2305.14627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14627"}},"official":{"repos":["princeton-nlp/alce"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lawyer-llama-technical-report","slug":"lawyer-llama-technical-report","title":"Lawyer LLaMA Technical Report","date":"2023-05-24","arxiv_id":"2305.15062","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lawyer-llama-technical-report#ran","syntology_url":"https://syntology.ai/paper/2305.15062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15062"}},"official":{"repos":["andrewzhe/lawyer-llama"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/wikichat-a-few-shot-llm-based-chatbot","slug":"wikichat-a-few-shot-llm-based-chatbot","title":"WikiChat: Stopping the Hallucination of Large Language Model Chatbots by Few-Shot Grounding on Wikipedia","date":"2023-05-23","arxiv_id":"2305.14292","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/wikichat-a-few-shot-llm-based-chatbot#ran","syntology_url":"https://syntology.ai/paper/2305.14292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14292"}},"official":{"repos":["stanford-oval/wikichat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-knowledge-a-framework-for-grounding","slug":"chain-of-knowledge-a-framework-for-grounding","title":"Chain-of-Knowledge: Grounding Large Language Models via Dynamic Knowledge Adapting over Heterogeneous Sources","date":"2023-05-22","arxiv_id":"2305.13269","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chain-of-knowledge-a-framework-for-grounding#ran","syntology_url":"https://syntology.ai/paper/2305.13269","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13269"}},"official":{"repos":["damo-nlp-sg/chain-of-knowledge"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/halomi-a-manually-annotated-benchmark-for","slug":"halomi-a-manually-annotated-benchmark-for","title":"HalOmi: A Manually Annotated Benchmark for Multilingual Hallucination and Omission Detection in Machine Translation","date":"2023-05-19","arxiv_id":"2305.11746","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/halomi-a-manually-annotated-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2305.11746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11746"}},"official":{"repos":["facebookresearch/stopes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/helma-a-large-scale-hallucination-evaluation","slug":"helma-a-large-scale-hallucination-evaluation","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","date":"2023-05-19","arxiv_id":"2305.11747","repositories_listed":3,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/helma-a-large-scale-hallucination-evaluation#ran","syntology_url":"https://syntology.ai/paper/2305.11747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11747"}},"official":{"repos":["RUCAIBox/HaluEval","rucaibox/helma"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-object-hallucination-in-large","slug":"evaluating-object-hallucination-in-large","title":"Evaluating Object Hallucination in Large Vision-Language Models","date":"2023-05-17","arxiv_id":"2305.10355","repositories_listed":6,"syntology":{"n":12,"n_ran":11,"n_constructed":2,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"11 ran (of which 2 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-object-hallucination-in-large#ran","syntology_url":"https://syntology.ai/paper/2305.10355","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10355"}},"official":{"repos":["rucaibox/pope"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/gpt-ner-named-entity-recognition-via-large","slug":"gpt-ner-named-entity-recognition-via-large","title":"GPT-NER: Named Entity Recognition via Large Language Models","date":"2023-04-20","arxiv_id":"2304.10428","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-ner-named-entity-recognition-via-large#ran","syntology_url":"https://syntology.ai/paper/2304.10428","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.10428"}},"official":{"repos":["shuhewang1998/gpt-ner"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/ovtrack-open-vocabulary-multiple-object","slug":"ovtrack-open-vocabulary-multiple-object","title":"OVTrack: Open-Vocabulary Multiple Object Tracking","date":"2023-04-17","arxiv_id":"2304.08408","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ovtrack-open-vocabulary-multiple-object#ran","syntology_url":"https://syntology.ai/paper/2304.08408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08408"}},"official":null}},{"url":"/paper/stereoscene-bev-assisted-stereo-matching","slug":"stereoscene-bev-assisted-stereo-matching","title":"Bridging Stereo Geometry and BEV Representation with Reliable Mutual Interaction for Semantic Scene Completion","date":"2023-03-24","arxiv_id":"2303.13959","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stereoscene-bev-assisted-stereo-matching#ran","syntology_url":"https://syntology.ai/paper/2303.13959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13959"}},"official":{"repos":["Arlo0o/StereoScene"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selfcheckgpt-zero-resource-black-box","slug":"selfcheckgpt-zero-resource-black-box","title":"SelfCheckGPT: Zero-Resource Black-Box Hallucination Detection for Generative Large Language Models","date":"2023-03-15","arxiv_id":"2303.08896","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/selfcheckgpt-zero-resource-black-box#ran","syntology_url":"https://syntology.ai/paper/2303.08896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08896"}},"official":{"repos":["potsawee/selfcheckgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-chain-of-thought-reasoning-in","slug":"multimodal-chain-of-thought-reasoning-in","title":"Multimodal Chain-of-Thought Reasoning in Language Models","date":"2023-02-02","arxiv_id":"2302.00923","repositories_listed":3,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multimodal-chain-of-thought-reasoning-in#ran","syntology_url":"https://syntology.ai/paper/2302.00923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00923"}},"official":{"repos":["amazon-science/mm-cot"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/mqag-multiple-choice-question-answering-and","slug":"mqag-multiple-choice-question-answering-and","title":"MQAG: Multiple-choice Question Answering and Generation for Assessing Information Consistency in Summarization","date":"2023-01-28","arxiv_id":"2301.12307","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mqag-multiple-choice-question-answering-and#ran","syntology_url":"https://syntology.ai/paper/2301.12307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12307"}},"official":{"repos":["potsawee/mqag0"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-and-detecting-hallucinations-in","slug":"understanding-and-detecting-hallucinations-in","title":"Understanding and Detecting Hallucinations in Neural Machine Translation via Model Introspection","date":"2023-01-18","arxiv_id":"2301.07779","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-and-detecting-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2301.07779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07779"}},"official":{"repos":["weijia-xu/hallucinations-in-nmt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interleaving-retrieval-with-chain-of-thought","slug":"interleaving-retrieval-with-chain-of-thought","title":"Interleaving Retrieval with Chain-of-Thought Reasoning for Knowledge-Intensive Multi-Step Questions","date":"2022-12-20","arxiv_id":"2212.10509","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interleaving-retrieval-with-chain-of-thought#ran","syntology_url":"https://syntology.ai/paper/2212.10509","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.10509"}},"official":{"repos":["stonybrooknlp/ircot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rho-r-reducing-hallucination-in-open-domain","slug":"rho-r-reducing-hallucination-in-open-domain","title":"RHO ($ρ$): Reducing Hallucination in Open-domain Dialogues with Knowledge Grounding","date":"2022-12-03","arxiv_id":"2212.01588","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rho-r-reducing-hallucination-in-open-domain#ran","syntology_url":"https://syntology.ai/paper/2212.01588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01588"}},"official":{"repos":["ziweiji/rho"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dataset-distillation-via-factorization","slug":"dataset-distillation-via-factorization","title":"Dataset Distillation via Factorization","date":"2022-10-30","arxiv_id":"2210.16774","repositories_listed":3,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/dataset-distillation-via-factorization#ran","syntology_url":"https://syntology.ai/paper/2210.16774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.16774"}},"official":{"repos":["huage001/datasetfactorization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/react-synergizing-reasoning-and-acting-in","slug":"react-synergizing-reasoning-and-acting-in","title":"ReAct: Synergizing Reasoning and Acting in Language Models","date":"2022-10-06","arxiv_id":"2210.03629","repositories_listed":9,"syntology":{"n":34,"n_ran":15,"n_constructed":9,"n_ran_checked":10,"n_instrument":5,"n_unverified":19,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":6,"phrase":"15 ran (of which 9 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 5 where Syntology's instrument failed) · 19 unverified","sample_list":"/paper/react-synergizing-reasoning-and-acting-in#ran","syntology_url":"https://syntology.ai/paper/2210.03629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03629"}},"official":{"repos":["ysymyth/ReAct"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/valhalla-visual-hallucination-for-machine","slug":"valhalla-visual-hallucination-for-machine","title":"VALHALLA: Visual Hallucination for Machine Translation","date":"2022-05-31","arxiv_id":"2206.00100","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":4,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/valhalla-visual-hallucination-for-machine#ran","syntology_url":"https://syntology.ai/paper/2206.00100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00100"}},"official":null}},{"url":"/paper/generating-natural-language-proofs-with","slug":"generating-natural-language-proofs-with","title":"Generating Natural Language Proofs with Verifier-Guided Search","date":"2022-05-25","arxiv_id":"2205.12443","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-natural-language-proofs-with#ran","syntology_url":"https://syntology.ai/paper/2205.12443","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12443"}},"official":{"repos":["princeton-nlp/NLProofS"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embedding-hallucination-for-few-shot-language","slug":"embedding-hallucination-for-few-shot-language","title":"Embedding Hallucination for Few-Shot Language Fine-tuning","date":"2022-05-03","arxiv_id":"2205.01307","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/embedding-hallucination-for-few-shot-language#ran","syntology_url":"https://syntology.ai/paper/2205.01307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01307"}},"official":{"repos":["yiren-jian/embedhalluc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/faithdial-a-faithful-benchmark-for","slug":"faithdial-a-faithful-benchmark-for","title":"FaithDial: A Faithful Benchmark for Information-Seeking Dialogue","date":"2022-04-22","arxiv_id":"2204.10757","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/faithdial-a-faithful-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2204.10757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.10757"}},"official":null}},{"url":"/paper/style-hallucinated-dual-consistency-learning","slug":"style-hallucinated-dual-consistency-learning","title":"Style-Hallucinated Dual Consistency Learning for Domain Generalized Semantic Segmentation","date":"2022-04-06","arxiv_id":"2204.02548","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/style-hallucinated-dual-consistency-learning#ran","syntology_url":"https://syntology.ai/paper/2204.02548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02548"}},"official":{"repos":["helioszhao/shade"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/towards-an-end-to-end-framework-for-flow","slug":"towards-an-end-to-end-framework-for-flow","title":"Towards An End-to-End Framework for Flow-Guided Video Inpainting","date":"2022-04-06","arxiv_id":"2204.02663","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-an-end-to-end-framework-for-flow#ran","syntology_url":"https://syntology.ai/paper/2204.02663","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02663"}},"official":{"repos":["MCG-NKU/E2FGVI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gcfsr-a-generative-and-controllable-face","slug":"gcfsr-a-generative-and-controllable-face","title":"GCFSR: a Generative and Controllable Face Super Resolution Method Without Facial and GAN Priors","date":"2022-03-14","arxiv_id":"2203.07319","repositories_listed":0,"syntology":{"n":20,"n_ran":11,"n_constructed":8,"n_ran_checked":10,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":20,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/gcfsr-a-generative-and-controllable-face#ran","syntology_url":"https://syntology.ai/paper/2203.07319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.07319"}},"official":null}},{"url":"/paper/hallucinated-neural-radiance-fields-in-the","slug":"hallucinated-neural-radiance-fields-in-the","title":"Hallucinated Neural Radiance Fields in the Wild","date":"2021-11-30","arxiv_id":"2111.15246","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallucinated-neural-radiance-fields-in-the#ran","syntology_url":"https://syntology.ai/paper/2111.15246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.15246"}},"official":null}},{"url":"/paper/let-there-be-a-clock-on-the-beach-reducing","slug":"let-there-be-a-clock-on-the-beach-reducing","title":"Let there be a clock on the beach: Reducing Object Hallucination in Image Captioning","date":"2021-10-04","arxiv_id":"2110.01705","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/let-there-be-a-clock-on-the-beach-reducing#ran","syntology_url":"https://syntology.ai/paper/2110.01705","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01705"}},"official":{"repos":["furkanbiten/object-bias"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-token-level-reference-free-hallucination","slug":"a-token-level-reference-free-hallucination","title":"A Token-level Reference-free Hallucination Detection Benchmark for Free-form Text Generation","date":"2021-04-18","arxiv_id":"2104.08704","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-token-level-reference-free-hallucination#ran","syntology_url":"https://syntology.ai/paper/2104.08704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.08704"}},"official":{"repos":["microsoft/HaDes"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/projected-distribution-loss-for-image","slug":"projected-distribution-loss-for-image","title":"Projected Distribution Loss for Image Enhancement","date":"2020-12-16","arxiv_id":"2012.09289","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/projected-distribution-loss-for-image#ran","syntology_url":"https://syntology.ai/paper/2012.09289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09289"}},"official":null}},{"url":"/paper/on-hallucinations-in-tomographic-image","slug":"on-hallucinations-in-tomographic-image","title":"On hallucinations in tomographic image reconstruction","date":"2020-12-01","arxiv_id":"2012.00646","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-hallucinations-in-tomographic-image#ran","syntology_url":"https://syntology.ai/paper/2012.00646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.00646"}},"official":null}},{"url":"/paper/3d-sketch-aware-semantic-scene-completion-via","slug":"3d-sketch-aware-semantic-scene-completion-via","title":"3D Sketch-aware Semantic Scene Completion via Semi-supervised Structure Prior","date":"2020-03-31","arxiv_id":"2003.14052","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-sketch-aware-semantic-scene-completion-via#ran","syntology_url":"https://syntology.ai/paper/2003.14052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.14052"}},"official":null}},{"url":"/paper/pulse-self-supervised-photo-upsampling-via","slug":"pulse-self-supervised-photo-upsampling-via","title":"PULSE: Self-Supervised Photo Upsampling via Latent Space Exploration of Generative Models","date":"2020-03-08","arxiv_id":"2003.03808","repositories_listed":16,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pulse-self-supervised-photo-upsampling-via#ran","syntology_url":"https://syntology.ai/paper/2003.03808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.03808"}},"official":{"repos":["adamian98/pulse"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/the-troublesome-kernel-why-deep-learning-for","slug":"the-troublesome-kernel-why-deep-learning-for","title":"The troublesome kernel -- On hallucinations, no free lunches and the accuracy-stability trade-off in inverse problems","date":"2020-01-05","arxiv_id":"2001.01258","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-troublesome-kernel-why-deep-learning-for#ran","syntology_url":"https://syntology.ai/paper/2001.01258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2001.01258"}},"official":{"repos":["vegarant/troublesome_kernel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"e050d41336007cf6b8c370407d5f8d47f580ac1b2f6f1f42a33a096c4fe7abe7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}