{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/7","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":19,"rows_per_page":100,"rows":[601,700],"of":1816,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination","prev":"/task/hallucination/papers/6","next":"/task/hallucination/papers/8","papers":[{"url":"/paper/btr-binary-token-representations-for","slug":"btr-binary-token-representations-for","title":"BTR: Binary Token Representations for Efficient Retrieval Augmented Language Models","date":"2023-10-02","arxiv_id":"2310.01329","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/btr-binary-token-representations-for#ran","syntology_url":"https://syntology.ai/paper/2310.01329","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01329"}},"official":{"repos":["csarron/btr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-lies-hallucinations-are-not-bugs-but","slug":"llm-lies-hallucinations-are-not-bugs-but","title":"LLM Lies: Hallucinations are not Bugs, but Features as Adversarial Examples","date":"2023-10-02","arxiv_id":"2310.01469","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-lies-hallucinations-are-not-bugs-but#ran","syntology_url":"https://syntology.ai/paper/2310.01469","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01469"}},"official":{"repos":["pku-yuangroup/hallucination-attack"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-and-mitigating-object-hallucination","slug":"analyzing-and-mitigating-object-hallucination","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","date":"2023-10-01","arxiv_id":"2310.00754","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/analyzing-and-mitigating-object-hallucination#ran","syntology_url":"https://syntology.ai/paper/2310.00754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00754"}},"official":{"repos":["yiyangzhou/lure"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/see-beyond-seeing-robust-3d-object-detection","slug":"see-beyond-seeing-robust-3d-object-detection","title":"Robust 3D Object Detection from LiDAR-Radar Point Clouds via Cross-Modal Feature Augmentation","date":"2023-09-29","arxiv_id":"2309.17336","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/see-beyond-seeing-robust-3d-object-detection#ran","syntology_url":"https://syntology.ai/paper/2309.17336","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17336"}},"official":{"repos":["djning/see_beyond_seeing"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-reduction-in-long-input-text","slug":"hallucination-reduction-in-long-input-text","title":"Hallucination Reduction in Long Input Text Summarization","date":"2023-09-28","arxiv_id":"2309.16781","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-cross-view-representation-1","slug":"self-supervised-cross-view-representation-1","title":"Self-supervised Cross-view Representation Reconstruction for Change Captioning","date":"2023-09-28","arxiv_id":"2309.16283","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/self-supervised-cross-view-representation-1#ran","syntology_url":"https://syntology.ai/paper/2309.16283","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16283"}},"official":{"repos":["tuyunbin/scorer"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lyra-orchestrating-dual-correction-in","slug":"lyra-orchestrating-dual-correction-in","title":"Lyra: Orchestrating Dual Correction in Automated Theorem Proving","date":"2023-09-27","arxiv_id":"2309.15806","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lyra-orchestrating-dual-correction-in#ran","syntology_url":"https://syntology.ai/paper/2309.15806","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.15806"}},"official":{"repos":["chuanyang-zheng/lyra-theorem-prover"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bamboo-a-comprehensive-benchmark-for","slug":"bamboo-a-comprehensive-benchmark-for","title":"BAMBOO: A Comprehensive Benchmark for Evaluating Long Text Modeling Capacities of Large Language Models","date":"2023-09-23","arxiv_id":"2309.13345","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/bamboo-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2309.13345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13345"}},"official":{"repos":["rucaibox/bamboo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-verification-reduces-hallucination","slug":"chain-of-verification-reduces-hallucination","title":"Chain-of-Verification Reduces Hallucination in Large Language Models","date":"2023-09-20","arxiv_id":"2309.11495","repositories_listed":1,"syntology":null},{"url":"/paper/pick-polished-informed-candidate-scoring-for","slug":"pick-polished-informed-candidate-scoring-for","title":"PICK: Polished & Informed Candidate Scoring for Knowledge-Grounded Dialogue Systems","date":"2023-09-19","arxiv_id":"2309.10413","repositories_listed":1,"syntology":null},{"url":"/paper/struc-bench-are-large-language-models-really","slug":"struc-bench-are-large-language-models-really","title":"Struc-Bench: Are Large Language Models Really Good at Generating Complex Structured Data?","date":"2023-09-16","arxiv_id":"2309.08963","repositories_listed":1,"syntology":null},{"url":"/paper/merge-conflicts-exploring-the-impacts-of","slug":"merge-conflicts-exploring-the-impacts-of","title":"\"Merge Conflicts!\" Exploring the Impacts of External Distractors to Parametric Knowledge Graphs","date":"2023-09-15","arxiv_id":"2309.08594","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"0 ran · 4 unverified","sample_list":"/paper/merge-conflicts-exploring-the-impacts-of#ran","syntology_url":"https://syntology.ai/paper/2309.08594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.08594"}},"official":{"repos":["qiancheng0/ekd_impacts_pkg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/cognitive-mirage-a-review-of-hallucinations","slug":"cognitive-mirage-a-review-of-hallucinations","title":"Cognitive Mirage: A Review of Hallucinations in Large Language Models","date":"2023-09-13","arxiv_id":"2309.06794","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-of-hallucination-in-large-foundation","slug":"a-survey-of-hallucination-in-large-foundation","title":"A Survey of Hallucination in Large Foundation Models","date":"2023-09-12","arxiv_id":"2309.05922","repositories_listed":1,"syntology":null},{"url":"/paper/tegit-generating-high-quality-instruction","slug":"tegit-generating-high-quality-instruction","title":"DoG-Instruct: Towards Premium Instruction-Tuning Data via Text-Grounded Instruction Wrapping","date":"2023-09-11","arxiv_id":"2309.05447","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tegit-generating-high-quality-instruction#ran","syntology_url":"https://syntology.ai/paper/2309.05447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.05447"}},"official":{"repos":["bahuia/dog-instruct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-tuning-large-language-models-with","slug":"knowledge-tuning-large-language-models-with","title":"Knowledge-tuning Large Language Models with Structured Medical Knowledge Bases for Reliable Response Generation in Chinese","date":"2023-09-08","arxiv_id":"2309.04175","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-tuning-large-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2309.04175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.04175"}},"official":null}},{"url":"/paper/zero-resource-hallucination-prevention-for","slug":"zero-resource-hallucination-prevention-for","title":"Zero-Resource Hallucination Prevention for Large Language Models","date":"2023-09-06","arxiv_id":"2309.02654","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-large-language-models-in","slug":"benchmarking-large-language-models-in","title":"Benchmarking Large Language Models in Retrieval-Augmented Generation","date":"2023-09-04","arxiv_id":"2309.01431","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2309.01431","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01431"}},"official":{"repos":["chen700564/RGB"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/siren-s-song-in-the-ai-ocean-a-survey-on","slug":"siren-s-song-in-the-ai-ocean-a-survey-on","title":"Siren's Song in the AI Ocean: A Survey on Hallucination in Large Language Models","date":"2023-09-03","arxiv_id":"2309.01219","repositories_listed":1,"syntology":null},{"url":"/paper/evaluation-and-analysis-of-hallucination-in","slug":"evaluation-and-analysis-of-hallucination-in","title":"Evaluation and Analysis of Hallucination in Large Vision-Language Models","date":"2023-08-29","arxiv_id":"2308.15126","repositories_listed":1,"syntology":null},{"url":"/paper/multi-party-goal-tracking-with-llms-comparing","slug":"multi-party-goal-tracking-with-llms-comparing","title":"Multi-party Goal Tracking with LLMs: Comparing Pre-training, Fine-tuning, and Prompt Engineering","date":"2023-08-29","arxiv_id":"2308.15231","repositories_listed":1,"syntology":null},{"url":"/paper/prefer-prompt-ensemble-learning-via-feedback","slug":"prefer-prompt-ensemble-learning-via-feedback","title":"PREFER: Prompt Ensemble Learning via Feedback-Reflect-Refine","date":"2023-08-23","arxiv_id":"2308.12033","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prefer-prompt-ensemble-learning-via-feedback#ran","syntology_url":"https://syntology.ai/paper/2308.12033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12033"}},"official":{"repos":["zcrwind/prefer"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/lan-hdr-luminance-based-alignment-network-for","slug":"lan-hdr-luminance-based-alignment-network-for","title":"LAN-HDR: Luminance-based Alignment Network for High Dynamic Range Video Reconstruction","date":"2023-08-22","arxiv_id":"2308.11116","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":1,"n_ran_checked":10,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":18,"phrase":"12 ran (of which 1 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/lan-hdr-luminance-based-alignment-network-for#ran","syntology_url":"https://syntology.ai/paper/2308.11116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.11116"}},"official":{"repos":["haesoochung/lan-hdr"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":1,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-on-wikipedia-style","slug":"large-language-models-on-wikipedia-style","title":"Large Language Models on Wikipedia-Style Survey Generation: an Evaluation in NLP Concepts","date":"2023-08-21","arxiv_id":"2308.10410","repositories_listed":1,"syntology":null},{"url":"/paper/mindmap-knowledge-graph-prompting-sparks","slug":"mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","arxiv_id":"2308.09729","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mindmap-knowledge-graph-prompting-sparks#ran","syntology_url":"https://syntology.ai/paper/2308.09729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09729"}},"official":{"repos":["wyl-willing/MindMap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-and-preventing-hallucinations-in","slug":"detecting-and-preventing-hallucinations-in","title":"Detecting and Preventing Hallucinations in Large Vision Language Models","date":"2023-08-11","arxiv_id":"2308.06394","repositories_listed":1,"syntology":null},{"url":"/paper/tiny-lvlm-ehub-early-multimodal-experiments","slug":"tiny-lvlm-ehub-early-multimodal-experiments","title":"TinyLVLM-eHub: Towards Comprehensive and Efficient Evaluation for Large Vision-Language Models","date":"2023-08-07","arxiv_id":"2308.03729","repositories_listed":1,"syntology":null},{"url":"/paper/automatically-correcting-large-language","slug":"automatically-correcting-large-language","title":"Automatically Correcting Large Language Models: Surveying the landscape of diverse self-correction strategies","date":"2023-08-06","arxiv_id":"2308.03188","repositories_listed":1,"syntology":null},{"url":"/paper/balanced-classification-a-unified-framework","slug":"balanced-classification-a-unified-framework","title":"Balanced Classification: A Unified Framework for Long-Tailed Object Detection","date":"2023-08-04","arxiv_id":"2308.02213","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-generic-enhancing-image-captioning","slug":"beyond-generic-enhancing-image-captioning","title":"Beyond Generic: Enhancing Image Captioning with Real-World Knowledge using Vision-Language Pre-Training Model","date":"2023-08-02","arxiv_id":"2308.01126","repositories_listed":1,"syntology":null},{"url":"/paper/transferable-decoding-with-visual-entities","slug":"transferable-decoding-with-visual-entities","title":"Transferable Decoding with Visual Entities for Zero-Shot Image Captioning","date":"2023-07-31","arxiv_id":"2307.16525","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":3,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/transferable-decoding-with-visual-entities#ran","syntology_url":"https://syntology.ai/paper/2307.16525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16525"}},"official":{"repos":["feielysia/viecap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/do-llms-possess-a-personality-making-the-mbti","slug":"do-llms-possess-a-personality-making-the-mbti","title":"Do LLMs Possess a Personality? Making the MBTI Test an Amazing Evaluation for Large Language Models","date":"2023-07-30","arxiv_id":"2307.16180","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/do-llms-possess-a-personality-making-the-mbti#ran","syntology_url":"https://syntology.ai/paper/2307.16180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16180"}},"official":{"repos":["harderthenharder/transformers_tasks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chatreport-democratizing-sustainability","slug":"chatreport-democratizing-sustainability","title":"CHATREPORT: Democratizing Sustainability Disclosure Analysis through LLM-based Tools","date":"2023-07-28","arxiv_id":"2307.15770","repositories_listed":1,"syntology":null},{"url":"/paper/med-halt-medical-domain-hallucination-test","slug":"med-halt-medical-domain-hallucination-test","title":"Med-HALT: Medical Domain Hallucination Test for Large Language Models","date":"2023-07-28","arxiv_id":"2307.15343","repositories_listed":1,"syntology":null},{"url":"/paper/pac-neural-prediction-set-learning-to","slug":"pac-neural-prediction-set-learning-to","title":"Selective Generation for Controllable Language Models","date":"2023-07-18","arxiv_id":"2307.09254","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pac-neural-prediction-set-learning-to#ran","syntology_url":"https://syntology.ai/paper/2307.09254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.09254"}},"official":{"repos":["ml-postech/selective-generation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/deficiency-aware-masked-transformer-for-video","slug":"deficiency-aware-masked-transformer-for-video","title":"Deficiency-Aware Masked Transformer for Video Inpainting","date":"2023-07-17","arxiv_id":"2307.08629","repositories_listed":1,"syntology":null},{"url":"/paper/prompts-should-not-be-seen-as-secrets","slug":"prompts-should-not-be-seen-as-secrets","title":"Effective Prompt Extraction from Language Models","date":"2023-07-13","arxiv_id":"2307.06865","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompts-should-not-be-seen-as-secrets#ran","syntology_url":"https://syntology.ai/paper/2307.06865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.06865"}},"official":{"repos":["y0mingzhang/prompt-extraction"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/parametric-depth-based-feature-representation","slug":"parametric-depth-based-feature-representation","title":"Parametric Depth Based Feature Representation Learning for Object Detection and Segmentation in Bird's Eye View","date":"2023-07-09","arxiv_id":"2307.04106","repositories_listed":1,"syntology":null},{"url":"/paper/chatlaw-open-source-legal-large-language","slug":"chatlaw-open-source-legal-large-language","title":"Chatlaw: A Multi-Agent Collaborative Legal Assistant with Knowledge Graph Enhanced Mixture-of-Experts Large Language Model","date":"2023-06-28","arxiv_id":"2306.16092","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-multimodal-large-language-models","slug":"a-survey-on-multimodal-large-language-models","title":"A Survey on Multimodal Large Language Models","date":"2023-06-23","arxiv_id":"2306.13549","repositories_listed":1,"syntology":null},{"url":"/paper/are-large-language-models-really-good-logical","slug":"are-large-language-models-really-good-logical","title":"Are Large Language Models Really Good Logical Reasoners? A Comprehensive Evaluation and Beyond","date":"2023-06-16","arxiv_id":"2306.09841","repositories_listed":1,"syntology":null},{"url":"/paper/kola-carefully-benchmarking-world-knowledge","slug":"kola-carefully-benchmarking-world-knowledge","title":"KoLA: Carefully Benchmarking World Knowledge of Large Language Models","date":"2023-06-15","arxiv_id":"2306.09296","repositories_listed":1,"syntology":null},{"url":"/paper/lvlm-ehub-a-comprehensive-evaluation","slug":"lvlm-ehub-a-comprehensive-evaluation","title":"LVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models","date":"2023-06-15","arxiv_id":"2306.09265","repositories_listed":1,"syntology":null},{"url":"/paper/aladdin-zero-shot-hallucination-of-stylized","slug":"aladdin-zero-shot-hallucination-of-stylized","title":"Aladdin: Zero-Shot Hallucination of Stylized 3D Assets from Abstract Scene Descriptions","date":"2023-06-09","arxiv_id":"2306.06212","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-interpretable-compressive-text","slug":"efficient-and-interpretable-compressive-text","title":"Efficient and Interpretable Compressive Text Summarisation with Unsupervised Dual-Agent Reinforcement Learning","date":"2023-06-06","arxiv_id":"2306.03415","repositories_listed":1,"syntology":null},{"url":"/paper/do-language-models-know-when-they-re","slug":"do-language-models-know-when-they-re","title":"Do Language Models Know When They're Hallucinating References?","date":"2023-05-29","arxiv_id":"2305.18248","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/do-language-models-know-when-they-re#ran","syntology_url":"https://syntology.ai/paper/2305.18248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18248"}},"official":{"repos":["microsoft/hallucinated-references"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-investigation-of-evaluation-metrics-for","slug":"an-investigation-of-evaluation-metrics-for","title":"An Investigation of Evaluation Metrics for Automated Medical Note Generation","date":"2023-05-27","arxiv_id":"2305.17364","repositories_listed":1,"syntology":null},{"url":"/paper/adaplanner-adaptive-planning-from-feedback-1","slug":"adaplanner-adaptive-planning-from-feedback-1","title":"AdaPlanner: Adaptive Planning from Feedback with Language Models","date":"2023-05-26","arxiv_id":"2305.16653","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaplanner-adaptive-planning-from-feedback-1#ran","syntology_url":"https://syntology.ai/paper/2305.16653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16653"}},"official":{"repos":["haotiansun14/adaplanner"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enabling-large-language-models-to-generate","slug":"enabling-large-language-models-to-generate","title":"Enabling Large Language Models to Generate Text with Citations","date":"2023-05-24","arxiv_id":"2305.14627","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enabling-large-language-models-to-generate#ran","syntology_url":"https://syntology.ai/paper/2305.14627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14627"}},"official":{"repos":["princeton-nlp/alce"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/getting-sick-after-seeing-a-doctor-diagnosing","slug":"getting-sick-after-seeing-a-doctor-diagnosing","title":"Getting Sick After Seeing a Doctor? Diagnosing and Mitigating Knowledge Conflicts in Event Temporal Reasoning","date":"2023-05-24","arxiv_id":"2305.14970","repositories_listed":1,"syntology":null},{"url":"/paper/gorilla-large-language-model-connected-with","slug":"gorilla-large-language-model-connected-with","title":"Gorilla: Large Language Model Connected with Massive APIs","date":"2023-05-24","arxiv_id":"2305.15334","repositories_listed":1,"syntology":null},{"url":"/paper/lawyer-llama-technical-report","slug":"lawyer-llama-technical-report","title":"Lawyer LLaMA Technical Report","date":"2023-05-24","arxiv_id":"2305.15062","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lawyer-llama-technical-report#ran","syntology_url":"https://syntology.ai/paper/2305.15062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15062"}},"official":{"repos":["andrewzhe/lawyer-llama"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/refgpt-reference-truthful-customized","slug":"refgpt-reference-truthful-customized","title":"RefGPT: Dialogue Generation of GPT, by GPT, and for GPT","date":"2023-05-24","arxiv_id":"2305.14994","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-language-model-hallucination-with","slug":"mitigating-language-model-hallucination-with","title":"The Knowledge Alignment Problem: Bridging Human and External Knowledge for Large Language Models","date":"2023-05-23","arxiv_id":"2305.13669","repositories_listed":1,"syntology":null},{"url":"/paper/pad-program-aided-distillation-specializes","slug":"pad-program-aided-distillation-specializes","title":"PaD: Program-aided Distillation Can Teach Small Models Reasoning Better than Chain-of-thought Fine-tuning","date":"2023-05-23","arxiv_id":"2305.13888","repositories_listed":1,"syntology":null},{"url":"/paper/sources-of-hallucination-by-large-language","slug":"sources-of-hallucination-by-large-language","title":"Sources of Hallucination by Large Language Models on Inference Tasks","date":"2023-05-23","arxiv_id":"2305.14552","repositories_listed":1,"syntology":null},{"url":"/paper/wikichat-a-few-shot-llm-based-chatbot","slug":"wikichat-a-few-shot-llm-based-chatbot","title":"WikiChat: Stopping the Hallucination of Large Language Model Chatbots by Few-Shot Grounding on Wikipedia","date":"2023-05-23","arxiv_id":"2305.14292","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/wikichat-a-few-shot-llm-based-chatbot#ran","syntology_url":"https://syntology.ai/paper/2305.14292","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14292"}},"official":{"repos":["stanford-oval/wikichat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-knowledge-a-framework-for-grounding","slug":"chain-of-knowledge-a-framework-for-grounding","title":"Chain-of-Knowledge: Grounding Large Language Models via Dynamic Knowledge Adapting over Heterogeneous Sources","date":"2023-05-22","arxiv_id":"2305.13269","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chain-of-knowledge-a-framework-for-grounding#ran","syntology_url":"https://syntology.ai/paper/2305.13269","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13269"}},"official":{"repos":["damo-nlp-sg/chain-of-knowledge"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/element-aware-summarization-with-large","slug":"element-aware-summarization-with-large","title":"Element-aware Summarization with Large Language Models: Expert-aligned Evaluation and Chain-of-Thought Method","date":"2023-05-22","arxiv_id":"2305.13412","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/element-aware-summarization-with-large#ran","syntology_url":"https://syntology.ai/paper/2305.13412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13412"}},"official":{"repos":["alsace08/sumcot"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/how-language-model-hallucinations-can","slug":"how-language-model-hallucinations-can","title":"How Language Model Hallucinations Can Snowball","date":"2023-05-22","arxiv_id":"2305.13534","repositories_listed":1,"syntology":null},{"url":"/paper/scene-graph-as-pivoting-inference-time-image","slug":"scene-graph-as-pivoting-inference-time-image","title":"Scene Graph as Pivoting: Inference-time Image-free Unsupervised Multimodal Machine Translation with Visual Scene Hallucination","date":"2023-05-20","arxiv_id":"2305.12256","repositories_listed":1,"syntology":null},{"url":"/paper/appraising-the-potential-uses-and-harms-of","slug":"appraising-the-potential-uses-and-harms-of","title":"Appraising the Potential Uses and Harms of LLMs for Medical Systematic Reviews","date":"2023-05-19","arxiv_id":"2305.11828","repositories_listed":1,"syntology":null},{"url":"/paper/halomi-a-manually-annotated-benchmark-for","slug":"halomi-a-manually-annotated-benchmark-for","title":"HalOmi: A Manually Annotated Benchmark for Multilingual Hallucination and Omission Detection in Machine Translation","date":"2023-05-19","arxiv_id":"2305.11746","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/halomi-a-manually-annotated-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2305.11746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11746"}},"official":{"repos":["facebookresearch/stopes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-chatgpt-a-good-causal-reasoner-a","slug":"is-chatgpt-a-good-causal-reasoner-a","title":"Is ChatGPT a Good Causal Reasoner? A Comprehensive Evaluation","date":"2023-05-12","arxiv_id":"2305.07375","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/is-chatgpt-a-good-causal-reasoner-a#ran","syntology_url":"https://syntology.ai/paper/2305.07375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07375"}},"official":{"repos":["ArrogantL/ChatGPT4CausalReasoning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/chartsumm-a-comprehensive-benchmark-for","slug":"chartsumm-a-comprehensive-benchmark-for","title":"ChartSumm: A Comprehensive Benchmark for Automatic Chart Summarization of Long and Short Summaries","date":"2023-04-26","arxiv_id":"2304.13620","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-chatgpt-4-on-acr-radiation","slug":"benchmarking-chatgpt-4-on-acr-radiation","title":"Benchmarking ChatGPT-4 on ACR Radiation Oncology In-Training (TXIT) Exam and Red Journal Gray Zone Cases: Potentials and Challenges for AI-Assisted Medical Education and Decision Making in Radiation Oncology","date":"2023-04-24","arxiv_id":"2304.11957","repositories_listed":1,"syntology":null},{"url":"/paper/ovtrack-open-vocabulary-multiple-object","slug":"ovtrack-open-vocabulary-multiple-object","title":"OVTrack: Open-Vocabulary Multiple Object Tracking","date":"2023-04-17","arxiv_id":"2304.08408","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ovtrack-open-vocabulary-multiple-object#ran","syntology_url":"https://syntology.ai/paper/2304.08408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08408"}},"official":null}},{"url":"/paper/stereoscene-bev-assisted-stereo-matching","slug":"stereoscene-bev-assisted-stereo-matching","title":"Bridging Stereo Geometry and BEV Representation with Reliable Mutual Interaction for Semantic Scene Completion","date":"2023-03-24","arxiv_id":"2303.13959","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stereoscene-bev-assisted-stereo-matching#ran","syntology_url":"https://syntology.ai/paper/2303.13959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13959"}},"official":{"repos":["Arlo0o/StereoScene"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/selfcheckgpt-zero-resource-black-box","slug":"selfcheckgpt-zero-resource-black-box","title":"SelfCheckGPT: Zero-Resource Black-Box Hallucination Detection for Generative Large Language Models","date":"2023-03-15","arxiv_id":"2303.08896","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/selfcheckgpt-zero-resource-black-box#ran","syntology_url":"https://syntology.ai/paper/2303.08896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08896"}},"official":{"repos":["potsawee/selfcheckgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/uprise-universal-prompt-retrieval-for","slug":"uprise-universal-prompt-retrieval-for","title":"UPRISE: Universal Prompt Retrieval for Improving Zero-Shot Evaluation","date":"2023-03-15","arxiv_id":"2303.08518","repositories_listed":1,"syntology":null},{"url":"/paper/halos-hallucination-free-organ-segmentation","slug":"halos-hallucination-free-organ-segmentation","title":"HALOS: Hallucination-free Organ Segmentation after Organ Resection Surgery","date":"2023-03-14","arxiv_id":"2303.07717","repositories_listed":1,"syntology":null},{"url":"/paper/a-multitask-multilingual-multimodal","slug":"a-multitask-multilingual-multimodal","title":"A Multitask, Multilingual, Multimodal Evaluation of ChatGPT on Reasoning, Hallucination, and Interactivity","date":"2023-02-08","arxiv_id":"2302.04023","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-and-detecting-hallucinations-in","slug":"understanding-and-detecting-hallucinations-in","title":"Understanding and Detecting Hallucinations in Neural Machine Translation via Model Introspection","date":"2023-01-18","arxiv_id":"2301.07779","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-and-detecting-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2301.07779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.07779"}},"official":{"repos":["weijia-xu/hallucinations-in-nmt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/diving-deep-into-modes-of-fact-hallucinations","slug":"diving-deep-into-modes-of-fact-hallucinations","title":"Diving Deep into Modes of Fact Hallucinations in Dialogue Systems","date":"2023-01-11","arxiv_id":"2301.04449","repositories_listed":1,"syntology":null},{"url":"/paper/doc2query-when-less-is-more","slug":"doc2query-when-less-is-more","title":"Doc2Query--: When Less is More","date":"2023-01-09","arxiv_id":"2301.03266","repositories_listed":1,"syntology":null},{"url":"/paper/you-truly-understand-what-i-need-intellectual","slug":"you-truly-understand-what-i-need-intellectual","title":"You Truly Understand What I Need: Intellectual and Friendly Dialogue Agents grounding Knowledge and Persona","date":"2023-01-06","arxiv_id":"2301.02401","repositories_listed":1,"syntology":null},{"url":"/paper/parametric-depth-based-feature-representation-1","slug":"parametric-depth-based-feature-representation-1","title":"Parametric Depth Based Feature Representation Learning for Object Detection and Segmentation in Bird's-Eye View","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-learning-reduces-hallucination-in","slug":"contrastive-learning-reduces-hallucination-in","title":"Contrastive Learning Reduces Hallucination in Conversations","date":"2022-12-20","arxiv_id":"2212.10400","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-transport-for-unsupervised-1","slug":"optimal-transport-for-unsupervised-1","title":"Optimal Transport for Unsupervised Hallucination Detection in Neural Machine Translation","date":"2022-12-19","arxiv_id":"2212.09631","repositories_listed":1,"syntology":null},{"url":"/paper/tokenization-consistency-matters-for","slug":"tokenization-consistency-matters-for","title":"Tokenization Consistency Matters for Generative Models on Extractive NLP Tasks","date":"2022-12-19","arxiv_id":"2212.09912","repositories_listed":1,"syntology":null},{"url":"/paper/style-hallucinated-dual-consistency-learning-1","slug":"style-hallucinated-dual-consistency-learning-1","title":"Style-Hallucinated Dual Consistency Learning: A Unified Framework for Visual Domain Generalization","date":"2022-12-18","arxiv_id":"2212.09068","repositories_listed":1,"syntology":null},{"url":"/paper/rho-r-reducing-hallucination-in-open-domain","slug":"rho-r-reducing-hallucination-in-open-domain","title":"RHO ($ρ$): Reducing Hallucination in Open-domain Dialogues with Knowledge Grounding","date":"2022-12-03","arxiv_id":"2212.01588","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rho-r-reducing-hallucination-in-open-domain#ran","syntology_url":"https://syntology.ai/paper/2212.01588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.01588"}},"official":{"repos":["ziweiji/rho"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-simultaneous-machine-translation","slug":"improving-simultaneous-machine-translation","title":"Improving Simultaneous Machine Translation with Monolingual Data","date":"2022-12-02","arxiv_id":"2212.01188","repositories_listed":1,"syntology":null},{"url":"/paper/dataset-factorization-for-condensation","slug":"dataset-factorization-for-condensation","title":"Dataset Factorization for Condensation","date":"2022-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/iterative-teaching-by-data-hallucination","slug":"iterative-teaching-by-data-hallucination","title":"Iterative Teaching by Data Hallucination","date":"2022-10-31","arxiv_id":"2210.17467","repositories_listed":1,"syntology":null},{"url":"/paper/ngep-a-graph-based-event-planning-framework","slug":"ngep-a-graph-based-event-planning-framework","title":"NGEP: A Graph-based Event Planning Framework for Story Generation","date":"2022-10-19","arxiv_id":"2210.10602","repositories_listed":1,"syntology":null},{"url":"/paper/normsage-multi-lingual-multi-cultural-norm","slug":"normsage-multi-lingual-multi-cultural-norm","title":"NormSAGE: Multi-Lingual Multi-Cultural Norm Discovery from Conversations On-the-Fly","date":"2022-10-16","arxiv_id":"2210.08604","repositories_listed":1,"syntology":null},{"url":"/paper/plausible-may-not-be-faithful-probing-object","slug":"plausible-may-not-be-faithful-probing-object","title":"Plausible May Not Be Faithful: Probing Object Hallucination in Vision-Language Pre-training","date":"2022-10-14","arxiv_id":"2210.07688","repositories_listed":1,"syntology":null},{"url":"/paper/thinking-hallucination-for-video-captioning","slug":"thinking-hallucination-for-video-captioning","title":"Thinking Hallucination for Video Captioning","date":"2022-09-28","arxiv_id":"2209.13853","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-task-specific-concept-knowledge","slug":"incorporating-task-specific-concept-knowledge","title":"Incorporating Task-specific Concept Knowledge into Script Learning","date":"2022-08-31","arxiv_id":"2209.00068","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-night-image-enhancement-when","slug":"unsupervised-night-image-enhancement-when","title":"Unsupervised Night Image Enhancement: When Layer Decomposition Meets Light-Effects Suppression","date":"2022-07-21","arxiv_id":"2207.10564","repositories_listed":1,"syntology":null},{"url":"/paper/reproducing-sensory-induced-hallucinations","slug":"reproducing-sensory-induced-hallucinations","title":"Reproducing sensory induced hallucinations via neural fields","date":"2022-07-08","arxiv_id":"2207.03901","repositories_listed":1,"syntology":null},{"url":"/paper/an-inflectional-database-for-gitksan","slug":"an-inflectional-database-for-gitksan","title":"An Inflectional Database for Gitksan","date":"2022-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/valhalla-visual-hallucination-for-machine","slug":"valhalla-visual-hallucination-for-machine","title":"VALHALLA: Visual Hallucination for Machine Translation","date":"2022-05-31","arxiv_id":"2206.00100","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":4,"n_pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 2 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/valhalla-visual-hallucination-for-machine#ran","syntology_url":"https://syntology.ai/paper/2206.00100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.00100"}},"official":null}},{"url":"/paper/learning-to-automate-follow-up-question","slug":"learning-to-automate-follow-up-question","title":"Learning to Automate Follow-up Question Generation using Process Knowledge for Depression Triage on Reddit Posts","date":"2022-05-27","arxiv_id":"2205.13884","repositories_listed":1,"syntology":null},{"url":"/paper/generating-natural-language-proofs-with","slug":"generating-natural-language-proofs-with","title":"Generating Natural Language Proofs with Verifier-Guided Search","date":"2022-05-25","arxiv_id":"2205.12443","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generating-natural-language-proofs-with#ran","syntology_url":"https://syntology.ai/paper/2205.12443","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.12443"}},"official":{"repos":["princeton-nlp/NLProofS"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embedding-hallucination-for-few-shot-language","slug":"embedding-hallucination-for-few-shot-language","title":"Embedding Hallucination for Few-Shot Language Fine-tuning","date":"2022-05-03","arxiv_id":"2205.01307","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/embedding-hallucination-for-few-shot-language#ran","syntology_url":"https://syntology.ai/paper/2205.01307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01307"}},"official":{"repos":["yiren-jian/embedhalluc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/faithdial-a-faithful-benchmark-for","slug":"faithdial-a-faithful-benchmark-for","title":"FaithDial: A Faithful Benchmark for Information-Seeking Dialogue","date":"2022-04-22","arxiv_id":"2204.10757","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/faithdial-a-faithful-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2204.10757","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.10757"}},"official":null}},{"url":"/paper/on-the-origin-of-hallucinations-in-1","slug":"on-the-origin-of-hallucinations-in-1","title":"On the Origin of Hallucinations in Conversational Models: Is it the Datasets or the Models?","date":"2022-04-17","arxiv_id":"2204.07931","repositories_listed":1,"syntology":null},{"url":"/paper/entity-driven-fact-aware-abstractive","slug":"entity-driven-fact-aware-abstractive","title":"Entity-driven Fact-aware Abstractive Summarization of Biomedical Literature","date":"2022-03-30","arxiv_id":"2203.15959","repositories_listed":1,"syntology":null}],"record_sha256":"2369817a9d74c11d3a791e246c923283bedee78e8636145f09baa502b0247e5c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}