{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/natural-language-inference/papers/3","list_of":"/task/natural-language-inference","task":"Natural Language Inference","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":20,"rows_per_page":100,"rows":[201,300],"of":1961,"counts":{"archive_papers_tagged":1961,"with_a_code_link":821,"where_syntology_ran_a_sample":209,"not_listed_spam_title":0,"listed":1961,"listed_where_code_ran":209,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":170,"every_run_a_failure_of_syntologys_instrument":39,"listed_with_a_run_with_no_instrument_failure":170,"listed_every_run_a_failure_of_syntologys_instrument":39,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/natural-language-inference","prev":"/task/natural-language-inference/papers/2","next":"/task/natural-language-inference/papers/4","papers":[{"url":"/paper/language-fusion-for-parameter-efficient-cross","slug":"language-fusion-for-parameter-efficient-cross","title":"Language Fusion for Parameter-Efficient Cross-lingual Transfer","date":"2025-01-12","arxiv_id":"2501.06892","repositories_listed":1,"syntology":null},{"url":"/paper/tougher-text-smarter-models-raising-the-bar","slug":"tougher-text-smarter-models-raising-the-bar","title":"Tougher Text, Smarter Models: Raising the Bar for Adversarial Defence Benchmarks","date":"2025-01-05","arxiv_id":"2501.02654","repositories_listed":1,"syntology":null},{"url":"/paper/defeasible-visual-entailment-benchmark","slug":"defeasible-visual-entailment-benchmark","title":"Defeasible Visual Entailment: Benchmark, Evaluator, and Reward-Driven Optimization","date":"2024-12-19","arxiv_id":"2412.16232","repositories_listed":1,"syntology":null},{"url":"/paper/on-adversarial-robustness-and-out-of","slug":"on-adversarial-robustness-and-out-of","title":"On Adversarial Robustness and Out-of-Distribution Robustness of Large Language Models","date":"2024-12-13","arxiv_id":"2412.10535","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/on-adversarial-robustness-and-out-of#ran","syntology_url":"https://syntology.ai/paper/2412.10535","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10535"}},"official":{"repos":["jordantab/llm-robustness-experiment"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bonafide-at-legallens-2024-shared-task-using","slug":"bonafide-at-legallens-2024-shared-task-using","title":"Bonafide at LegalLens 2024 Shared Task: Using Lightweight DeBERTa Based Encoder For Legal Violation Detection and Resolution","date":"2024-10-30","arxiv_id":"2410.22977","repositories_listed":1,"syntology":null},{"url":"/paper/flexible-natural-language-based-image-data","slug":"flexible-natural-language-based-image-data","title":"Flexible Natural Language-Based Image Data Downlink Prioritization for Nanosatellites","date":"2024-10-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/uottawa-at-legallens-2024-transformer-based","slug":"uottawa-at-legallens-2024-transformer-based","title":"uOttawa at LegalLens-2024: Transformer-based Classification Experiments","date":"2024-10-28","arxiv_id":"2410.21139","repositories_listed":1,"syntology":null},{"url":"/paper/augmenting-legal-decision-support-systems","slug":"augmenting-legal-decision-support-systems","title":"Augmenting Legal Decision Support Systems with LLM-based NLI for Analyzing Social Media Evidence","date":"2024-10-21","arxiv_id":"2410.15990","repositories_listed":1,"syntology":null},{"url":"/paper/inference-and-verbalization-functions-during","slug":"inference-and-verbalization-functions-during","title":"Inference and Verbalization Functions During In-Context Learning","date":"2024-10-12","arxiv_id":"2410.09349","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inference-and-verbalization-functions-during#ran","syntology_url":"https://syntology.ai/paper/2410.09349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09349"}},"official":{"repos":["junyitao/infer-then-verbalize-during-icl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/veritas-nli-validation-and-extraction-of","slug":"veritas-nli-validation-and-extraction-of","title":"VERITAS-NLI : Validation and Extraction of Reliable Information Through Automated Scraping and Natural Language Inference","date":"2024-10-12","arxiv_id":"2410.09455","repositories_listed":1,"syntology":null},{"url":"/paper/dadee-unsupervised-domain-adaptation-in-early","slug":"dadee-unsupervised-domain-adaptation-in-early","title":"DAdEE: Unsupervised Domain Adaptation in Early Exit PLMs","date":"2024-10-06","arxiv_id":"2410.04424","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dadee-unsupervised-domain-adaptation-in-early#ran","syntology_url":"https://syntology.ai/paper/2410.04424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04424"}},"official":{"repos":["div290/dadee"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/econ-on-the-detection-and-resolution-of","slug":"econ-on-the-detection-and-resolution-of","title":"ECon: On the Detection and Resolution of Evidence Conflicts","date":"2024-10-05","arxiv_id":"2410.04068","repositories_listed":1,"syntology":null},{"url":"/paper/take-it-easy-label-adaptive-self","slug":"take-it-easy-label-adaptive-self","title":"Take It Easy: Label-Adaptive Self-Rationalization for Fact Verification and Explanation Generation","date":"2024-10-05","arxiv_id":"2410.04002","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/take-it-easy-label-adaptive-self#ran","syntology_url":"https://syntology.ai/paper/2410.04002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04002"}},"official":{"repos":["jingyng/label-adaptive-self-rationalization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-hard-is-this-test-set-nli","slug":"how-hard-is-this-test-set-nli","title":"How Hard is this Test Set? NLI Characterization by Exploiting Training Dynamics","date":"2024-10-04","arxiv_id":"2410.03429","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-coherent-explanation-generation-of","slug":"multimodal-coherent-explanation-generation-of","title":"Multimodal Coherent Explanation Generation of Robot Failures","date":"2024-10-01","arxiv_id":"2410.00659","repositories_listed":1,"syntology":null},{"url":"/paper/assessment-and-manipulation-of-latent","slug":"assessment-and-manipulation-of-latent","title":"Assessment and manipulation of latent constructs in pre-trained language models using psychometric scales","date":"2024-09-29","arxiv_id":"2409.19655","repositories_listed":1,"syntology":null},{"url":"/paper/disgem-distractor-generation-for-multiple","slug":"disgem-distractor-generation-for-multiple","title":"DisGeM: Distractor Generation for Multiple Choice Questions with Span Masking","date":"2024-09-26","arxiv_id":"2409.18263","repositories_listed":1,"syntology":null},{"url":"/paper/lowrem-a-repository-of-word-embeddings-for-87","slug":"lowrem-a-repository-of-word-embeddings-for-87","title":"GrEmLIn: A Repository of Green Baseline Embeddings for 87 Low-Resource Languages Injected with Multilingual Graph Knowledge","date":"2024-09-26","arxiv_id":"2409.18193","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-adversarial-robustness-in-natural","slug":"enhancing-adversarial-robustness-in-natural","title":"Enhancing adversarial robustness in Natural Language Inference using explanations","date":"2024-09-11","arxiv_id":"2409.07423","repositories_listed":1,"syntology":null},{"url":"/paper/application-specific-compression-of-deep","slug":"application-specific-compression-of-deep","title":"Application Specific Compression of Deep Learning Models","date":"2024-09-09","arxiv_id":"2409.05368","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-pre-training-and-self","slug":"a-comparative-study-of-pre-training-and-self","title":"A Comparative Study of Pre-training and Self-training","date":"2024-09-04","arxiv_id":"2409.02751","repositories_listed":1,"syntology":null},{"url":"/paper/concse-unified-contrastive-learning-and","slug":"concse-unified-contrastive-learning-and","title":"ConCSE: Unified Contrastive Learning and Augmentation for Code-Switched Embeddings","date":"2024-08-28","arxiv_id":"2409.00120","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-factual-consistency-evaluation","slug":"zero-shot-factual-consistency-evaluation","title":"Zero-shot Factual Consistency Evaluation Across Domains","date":"2024-08-07","arxiv_id":"2408.04114","repositories_listed":1,"syntology":null},{"url":"/paper/2408-03127","slug":"2408-03127","title":"Lisbon Computational Linguists at SemEval-2024 Task 2: Using A Mistral 7B Model and Data Augmentation","date":"2024-08-06","arxiv_id":"2408.03127","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00675","slug":"2408-00675","title":"Leveraging Entailment Judgements in Cross-Lingual Summarisation","date":"2024-08-01","arxiv_id":"2408.00675","repositories_listed":1,"syntology":null},{"url":"/paper/scientific-qa-system-with-verifiable-answers","slug":"scientific-qa-system-with-verifiable-answers","title":"Scientific QA System with Verifiable Answers","date":"2024-07-16","arxiv_id":"2407.11485","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-zero-shot-crosslingual-performance","slug":"boosting-zero-shot-crosslingual-performance","title":"Boosting Zero-Shot Crosslingual Performance using LLM-Based Augmentations with Effective Data Selection","date":"2024-07-15","arxiv_id":"2407.10582","repositories_listed":1,"syntology":null},{"url":"/paper/farfetched-entity-centric-reasoning-and-claim-1","slug":"farfetched-entity-centric-reasoning-and-claim-1","title":"FarFetched: Entity-centric Reasoning and Claim Validation for the Greek Language based on Textually Represented Environments","date":"2024-07-13","arxiv_id":"2407.09888","repositories_listed":1,"syntology":null},{"url":"/paper/anah-v2-scaling-analytical-hallucination","slug":"anah-v2-scaling-analytical-hallucination","title":"ANAH-v2: Scaling Analytical Hallucination Annotation of Large Language Models","date":"2024-07-05","arxiv_id":"2407.04693","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/anah-v2-scaling-analytical-hallucination#ran","syntology_url":"https://syntology.ai/paper/2407.04693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04693"}},"official":{"repos":["open-compass/anah"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/extracting-and-encoding-leveraging-large","slug":"extracting-and-encoding-leveraging-large","title":"Extracting and Encoding: Leveraging Large Language Models and Medical Knowledge to Enhance Radiological Text Representation","date":"2024-07-02","arxiv_id":"2407.01948","repositories_listed":1,"syntology":null},{"url":"/paper/econnli-evaluating-large-language-models-on","slug":"econnli-evaluating-large-language-models-on","title":"EconNLI: Evaluating Large Language Models on Economics Reasoning","date":"2024-07-01","arxiv_id":"2407.01212","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/econnli-evaluating-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2407.01212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01212"}},"official":{"repos":["irenehere/econnli"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/seeing-the-big-through-the-small-can-llms","slug":"seeing-the-big-through-the-small-can-llms","title":"\"Seeing the Big through the Small\": Can LLMs Approximate Human Judgment Distributions on NLI from a Few Explanations?","date":"2024-06-25","arxiv_id":"2406.17600","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seeing-the-big-through-the-small-can-llms#ran","syntology_url":"https://syntology.ai/paper/2406.17600","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17600"}},"official":{"repos":["mainlp/mjd-estimator"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/co-training-for-low-resource-scientific","slug":"co-training-for-low-resource-scientific","title":"Co-training for Low Resource Scientific Natural Language Inference","date":"2024-06-20","arxiv_id":"2406.14666","repositories_listed":1,"syntology":null},{"url":"/paper/fzi-wim-at-semeval-2024-task-2-self","slug":"fzi-wim-at-semeval-2024-task-2-self","title":"FZI-WIM at SemEval-2024 Task 2: Self-Consistent CoT for Complex NLI in Biomedical Domain","date":"2024-06-14","arxiv_id":"2406.10040","repositories_listed":1,"syntology":null},{"url":"/paper/do-language-models-understand-morality","slug":"do-language-models-understand-morality","title":"Do Language Models Understand Morality? Towards a Robust Detection of Moral Content","date":"2024-06-06","arxiv_id":"2406.04143","repositories_listed":1,"syntology":null},{"url":"/paper/css-contrastive-semantic-similarity-for","slug":"css-contrastive-semantic-similarity-for","title":"CSS: Contrastive Semantic Similarity for Uncertainty Quantification of LLMs","date":"2024-06-05","arxiv_id":"2406.03158","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/css-contrastive-semantic-similarity-for#ran","syntology_url":"https://syntology.ai/paper/2406.03158","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03158"}},"official":{"repos":["aoshuang92/css_uq_llms"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/accurate-and-nuanced-open-qa-evaluation","slug":"accurate-and-nuanced-open-qa-evaluation","title":"Accurate and Nuanced Open-QA Evaluation Through Textual Entailment","date":"2024-05-26","arxiv_id":"2405.16702","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accurate-and-nuanced-open-qa-evaluation#ran","syntology_url":"https://syntology.ai/paper/2405.16702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16702"}},"official":{"repos":["U-Alberta/QA-partial-marks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-novel-cartography-based-curriculum-learning","slug":"a-novel-cartography-based-curriculum-learning","title":"A Novel Cartography-Based Curriculum Learning Method Applied on RoNLI: The First Romanian Natural Language Inference Corpus","date":"2024-05-20","arxiv_id":"2405.11877","repositories_listed":1,"syntology":null},{"url":"/paper/from-text-to-context-an-entailment-approach","slug":"from-text-to-context-an-entailment-approach","title":"From Text to Context: An Entailment Approach for News Stakeholder Classification","date":"2024-05-14","arxiv_id":"2405.08751","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-and-optimizing-global","slug":"quantifying-and-optimizing-global","title":"Quantifying and Optimizing Global Faithfulness in Persona-driven Role-playing","date":"2024-05-13","arxiv_id":"2405.07726","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/quantifying-and-optimizing-global#ran","syntology_url":"https://syntology.ai/paper/2405.07726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.07726"}},"official":{"repos":["KomeijiForce/Active_Passive_Constraint_Koishiday_2024"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-statements-in-text-a-domain","slug":"detecting-statements-in-text-a-domain","title":"Detecting Statements in Text: A Domain-Agnostic Few-Shot Solution","date":"2024-05-09","arxiv_id":"2405.05705","repositories_listed":1,"syntology":null},{"url":"/paper/the-effect-of-model-size-on-llm-post-hoc","slug":"the-effect-of-model-size-on-llm-post-hoc","title":"The Effect of Model Size on LLM Post-hoc Explainability via LIME","date":"2024-05-08","arxiv_id":"2405.05348","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-effect-of-model-size-on-llm-post-hoc#ran","syntology_url":"https://syntology.ai/paper/2405.05348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05348"}},"official":{"repos":["henningheyen/scalability-of-llm-posthoc-explanations"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/d-nlp-at-semeval-2024-task-2-evaluating","slug":"d-nlp-at-semeval-2024-task-2-evaluating","title":"D-NLP at SemEval-2024 Task 2: Evaluating Clinical Inference Capabilities of Large Language Models","date":"2024-05-07","arxiv_id":"2405.04170","repositories_listed":1,"syntology":null},{"url":"/paper/unraveling-the-dominance-of-large-language","slug":"unraveling-the-dominance-of-large-language","title":"Unraveling the Dominance of Large Language Models Over Transformer Models for Bangla Natural Language Inference: A Comprehensive Study","date":"2024-05-05","arxiv_id":"2405.02937","repositories_listed":1,"syntology":null},{"url":"/paper/verification-and-refinement-of-natural","slug":"verification-and-refinement-of-natural","title":"Verification and Refinement of Natural Language Explanations through LLM-Symbolic Theorem Proving","date":"2024-05-02","arxiv_id":"2405.01379","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/verification-and-refinement-of-natural#ran","syntology_url":"https://syntology.ai/paper/2405.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01379"}},"official":{"repos":["neuro-symbolic-ai/explanation_refinement"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncovering-agendas-a-novel-french-english","slug":"uncovering-agendas-a-novel-french-english","title":"Uncovering Agendas: A Novel French & English Dataset for Agenda Detection on Social Media","date":"2024-05-01","arxiv_id":"2405.00821","repositories_listed":1,"syntology":null},{"url":"/paper/don-t-say-no-jailbreaking-llm-by-suppressing","slug":"don-t-say-no-jailbreaking-llm-by-suppressing","title":"Don't Say No: Jailbreaking LLM by Suppressing Refusal","date":"2024-04-25","arxiv_id":"2404.16369","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/don-t-say-no-jailbreaking-llm-by-suppressing#ran","syntology_url":"https://syntology.ai/paper/2404.16369","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16369"}},"official":{"repos":["dsn-2024/dsn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/caspr-automated-evaluation-metric-for","slug":"caspr-automated-evaluation-metric-for","title":"CASPR: Automated Evaluation Metric for Contrastive Summarization","date":"2024-04-23","arxiv_id":"2404.15565","repositories_listed":1,"syntology":null},{"url":"/paper/automated-long-answer-grading-with-ricechem","slug":"automated-long-answer-grading-with-ricechem","title":"Automated Long Answer Grading with RiceChem Dataset","date":"2024-04-22","arxiv_id":"2404.14316","repositories_listed":1,"syntology":null},{"url":"/paper/tldr-at-semeval-2024-task-2-t5-generated","slug":"tldr-at-semeval-2024-task-2-t5-generated","title":"TLDR at SemEval-2024 Task 2: T5-generated clinical-Language summaries for DeBERTa Report Analysis","date":"2024-04-14","arxiv_id":"2404.09136","repositories_listed":1,"syntology":null},{"url":"/paper/mscinli-a-diverse-benchmark-for-scientific","slug":"mscinli-a-diverse-benchmark-for-scientific","title":"MSciNLI: A Diverse Benchmark for Scientific Natural Language Inference","date":"2024-04-11","arxiv_id":"2404.08066","repositories_listed":1,"syntology":null},{"url":"/paper/multi-image-visual-question-answering-for","slug":"multi-image-visual-question-answering-for","title":"Language Models Meet Anomaly Detection for Better Interpretability and Generalizability","date":"2024-04-11","arxiv_id":"2404.07622","repositories_listed":1,"syntology":null},{"url":"/paper/xnlieu-a-dataset-for-cross-lingual-nli-in","slug":"xnlieu-a-dataset-for-cross-lingual-nli-in","title":"XNLIeu: a dataset for cross-lingual NLI in Basque","date":"2024-04-10","arxiv_id":"2404.06996","repositories_listed":1,"syntology":null},{"url":"/paper/iitk-at-semeval-2024-task-2-exploring-the","slug":"iitk-at-semeval-2024-task-2-exploring-the","title":"IITK at SemEval-2024 Task 2: Exploring the Capabilities of LLMs for Safe Biomedical Natural Language Inference for Clinical Trials","date":"2024-04-06","arxiv_id":"2404.04510","repositories_listed":1,"syntology":null},{"url":"/paper/forget-nli-use-a-dictionary-zero-shot-topic","slug":"forget-nli-use-a-dictionary-zero-shot-topic","title":"Forget NLI, Use a Dictionary: Zero-Shot Topic Classification for Low-Resource Languages with Application to Luxembourgish","date":"2024-04-05","arxiv_id":"2404.03912","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-robustness-of-modelling","slug":"investigating-the-robustness-of-modelling","title":"Investigating the Robustness of Modelling Decisions for Few-Shot Cross-Topic Stance Detection: A Preregistered Study","date":"2024-04-05","arxiv_id":"2404.03987","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-generative-language-models-in","slug":"evaluating-generative-language-models-in","title":"Evaluating Generative Language Models in Information Extraction as Subjective Question Correction","date":"2024-04-04","arxiv_id":"2404.03532","repositories_listed":1,"syntology":null},{"url":"/paper/affective-nli-towards-accurate-and","slug":"affective-nli-towards-accurate-and","title":"Affective-NLI: Towards Accurate and Interpretable Personality Recognition in Conversation","date":"2024-04-03","arxiv_id":"2404.02589","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-role-of-summary-content-units-in-text","slug":"on-the-role-of-summary-content-units-in-text","title":"On the Role of Summary Content Units in Text Summarization Evaluation","date":"2024-04-02","arxiv_id":"2404.01701","repositories_listed":1,"syntology":null},{"url":"/paper/ails-ntua-at-semeval-2024-task-6-efficient","slug":"ails-ntua-at-semeval-2024-task-6-efficient","title":"AILS-NTUA at SemEval-2024 Task 6: Efficient model tuning for hallucination detection and analysis","date":"2024-04-01","arxiv_id":"2404.01210","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-divergent-inductive-biases-of-llms","slug":"unveiling-divergent-inductive-biases-of-llms","title":"Unveiling Divergent Inductive Biases of LLMs on Temporal Data","date":"2024-04-01","arxiv_id":"2404.01453","repositories_listed":1,"syntology":null},{"url":"/paper/edinburgh-clinical-nlp-at-semeval-2024-task-2","slug":"edinburgh-clinical-nlp-at-semeval-2024-task-2","title":"Edinburgh Clinical NLP at SemEval-2024 Task 2: Fine-tune your model unless you have access to GPT-4","date":"2024-03-30","arxiv_id":"2404.00484","repositories_listed":1,"syntology":null},{"url":"/paper/adverb-is-the-key-simple-text-data","slug":"adverb-is-the-key-simple-text-data","title":"Adverb Is the Key: Simple Text Data Augmentation with Adverb Deletion","date":"2024-03-29","arxiv_id":"2403.20015","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adverb-is-the-key-simple-text-data#ran","syntology_url":"https://syntology.ai/paper/2403.20015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20015"}},"official":{"repos":["c-juhwan/adverb-deletion-aug"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-modularity-transferable-a-case-study","slug":"is-modularity-transferable-a-case-study","title":"Is Modularity Transferable? A Case Study through the Lens of Knowledge Distillation","date":"2024-03-27","arxiv_id":"2403.18804","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-tokenization-strategies-and","slug":"exploring-tokenization-strategies-and","title":"Exploring Tokenization Strategies and Vocabulary Sizes for Enhanced Arabic Language Models","date":"2024-03-17","arxiv_id":"2403.11130","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-continual-learning-of-compositional","slug":"exploring-continual-learning-of-compositional","title":"Exploring Continual Learning of Compositional Generalization in NLI","date":"2024-03-07","arxiv_id":"2403.04400","repositories_listed":1,"syntology":null},{"url":"/paper/fenice-factuality-evaluation-of-summarization","slug":"fenice-factuality-evaluation-of-summarization","title":"FENICE: Factuality Evaluation of summarization based on Natural language Inference and Claim Extraction","date":"2024-03-04","arxiv_id":"2403.02270","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-use-of-silver-standard-data-for-zero","slug":"on-the-use-of-silver-standard-data-for-zero","title":"On the use of Silver Standard Data for Zero-shot Classification Tasks in Information Extraction","date":"2024-02-28","arxiv_id":"2402.18061","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-natural-language-inference-based","slug":"fine-grained-natural-language-inference-based","title":"Fine-Grained Natural Language Inference Based Faithfulness Evaluation for Diverse Summarisation Tasks","date":"2024-02-27","arxiv_id":"2402.17630","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-natural-language-inference-based#ran","syntology_url":"https://syntology.ai/paper/2402.17630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17630"}},"official":{"repos":["hjznlp/infuse"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citation-enhanced-generation-for-llm-based","slug":"citation-enhanced-generation-for-llm-based","title":"Citation-Enhanced Generation for LLM-based Chatbots","date":"2024-02-25","arxiv_id":"2402.16063","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-humans-write-code-large-models-do-it","slug":"how-do-humans-write-code-large-models-do-it","title":"How Do Humans Write Code? Large Models Do It the Same Way Too","date":"2024-02-24","arxiv_id":"2402.15729","repositories_listed":1,"syntology":null},{"url":"/paper/gpt-hatecheck-can-llms-write-better","slug":"gpt-hatecheck-can-llms-write-better","title":"GPT-HateCheck: Can LLMs Write Better Functional Tests for Hate Speech Detection?","date":"2024-02-23","arxiv_id":"2402.15238","repositories_listed":1,"syntology":null},{"url":"/paper/improving-sentence-embeddings-with-an","slug":"improving-sentence-embeddings-with-an","title":"Improving Sentence Embeddings with Automatic Generation of Training Data Using Few-shot Examples","date":"2024-02-23","arxiv_id":"2402.15132","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-sentence-embeddings-with-an#ran","syntology_url":"https://syntology.ai/paper/2402.15132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15132"}},"official":{"repos":["lamsoma/auto_nli"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conformalized-credal-set-predictors","slug":"conformalized-credal-set-predictors","title":"Conformalized Credal Set Predictors","date":"2024-02-16","arxiv_id":"2402.10723","repositories_listed":1,"syntology":null},{"url":"/paper/plausible-extractive-rationalization-through","slug":"plausible-extractive-rationalization-through","title":"Plausible Extractive Rationalization through Semi-Supervised Entailment Signal","date":"2024-02-13","arxiv_id":"2402.08479","repositories_listed":1,"syntology":null},{"url":"/paper/a-hypothesis-driven-framework-for-the","slug":"a-hypothesis-driven-framework-for-the","title":"A Hypothesis-Driven Framework for the Analysis of Self-Rationalising Models","date":"2024-02-07","arxiv_id":"2402.04787","repositories_listed":1,"syntology":null},{"url":"/paper/hqa-attack-toward-high-quality-black-box-hard-1","slug":"hqa-attack-toward-high-quality-black-box-hard-1","title":"HQA-Attack: Toward High Quality Black-Box Hard-Label Adversarial Attack on Text","date":"2024-02-02","arxiv_id":"2402.01806","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-ethical-explanations-of-large","slug":"enhancing-ethical-explanations-of-large","title":"Enhancing Ethical Explanations of Large Language Models through Iterative Symbolic Refinement","date":"2024-02-01","arxiv_id":"2402.00745","repositories_listed":1,"syntology":null},{"url":"/paper/mt-ranker-reference-free-machine-translation","slug":"mt-ranker-reference-free-machine-translation","title":"MT-Ranker: Reference-free machine translation evaluation by inter-system ranking","date":"2024-01-30","arxiv_id":"2401.17099","repositories_listed":1,"syntology":null},{"url":"/paper/infolossqa-characterizing-and-recovering","slug":"infolossqa-characterizing-and-recovering","title":"InfoLossQA: Characterizing and Recovering Information Loss in Text Simplification","date":"2024-01-29","arxiv_id":"2401.16475","repositories_listed":1,"syntology":null},{"url":"/paper/textual-entailment-for-effective-triple","slug":"textual-entailment-for-effective-triple","title":"Textual Entailment for Effective Triple Validation in Object Prediction","date":"2024-01-29","arxiv_id":"2401.16293","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-sensitivities-and-inconsistent","slug":"semantic-sensitivities-and-inconsistent","title":"Semantic Sensitivities and Inconsistent Predictions: Measuring the Fragility of NLI Models","date":"2024-01-25","arxiv_id":"2401.14440","repositories_listed":1,"syntology":null},{"url":"/paper/seed-guided-fine-grained-entity-typing-in","slug":"seed-guided-fine-grained-entity-typing-in","title":"Seed-Guided Fine-Grained Entity Typing in Science and Engineering Domains","date":"2024-01-23","arxiv_id":"2401.13129","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/seed-guided-fine-grained-entity-typing-in#ran","syntology_url":"https://syntology.ai/paper/2401.13129","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13129"}},"official":{"repos":["yuzhimanhua/setype"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/can-large-language-models-explain-themselves-1","slug":"can-large-language-models-explain-themselves-1","title":"Are self-explanations from Large Language Models faithful?","date":"2024-01-15","arxiv_id":"2401.07927","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-large-language-models-explain-themselves-1#ran","syntology_url":"https://syntology.ai/paper/2401.07927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07927"}},"official":{"repos":["AndreasMadsen/llm-introspection"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-editing-can-hurt-general-abilities-of","slug":"model-editing-can-hurt-general-abilities-of","title":"Model Editing Harms General Abilities of Large Language Models: Regularization to the Rescue","date":"2024-01-09","arxiv_id":"2401.04700","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/model-editing-can-hurt-general-abilities-of#ran","syntology_url":"https://syntology.ai/paper/2401.04700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04700"}},"official":{"repos":["jasonforjoy/model-editing-hurt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-convergence-of-semi-unsupervised","slug":"on-the-convergence-of-semi-unsupervised","title":"On the Stability of a non-hyperbolic nonlinear map with non-bounded set of non-isolated fixed points with applications to Machine Learning","date":"2024-01-05","arxiv_id":"2401.03051","repositories_listed":1,"syntology":null},{"url":"/paper/building-efficient-universal-classifiers-with","slug":"building-efficient-universal-classifiers-with","title":"Building Efficient Universal Classifiers with Natural Language Inference","date":"2023-12-29","arxiv_id":"2312.17543","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/building-efficient-universal-classifiers-with#ran","syntology_url":"https://syntology.ai/paper/2312.17543","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17543"}},"official":{"repos":["moritzlaurer/zeroshot-classifier"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/l-tuning-synchronized-label-tuning-for-prompt","slug":"l-tuning-synchronized-label-tuning-for-prompt","title":"L-TUNING: Synchronized Label Tuning for Prompt and Prefix in LLMs","date":"2023-12-21","arxiv_id":"2402.01643","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-highly-influential-shortcut","slug":"discovering-highly-influential-shortcut","title":"Discovering Highly Influential Shortcut Reasoning: An Automated Template-Free Approach","date":"2023-12-15","arxiv_id":"2312.09718","repositories_listed":1,"syntology":null},{"url":"/paper/dissecting-vocabulary-biases-datasets-through","slug":"dissecting-vocabulary-biases-datasets-through","title":"Dissecting vocabulary biases datasets through statistical testing and automated data augmentation for artifact mitigation in Natural Language Inference","date":"2023-12-14","arxiv_id":"2312.08747","repositories_listed":1,"syntology":null},{"url":"/paper/quantifying-divergence-for-human-ai","slug":"quantifying-divergence-for-human-ai","title":"Quantifying Divergence for Human-AI Collaboration and Cognitive Trust","date":"2023-12-14","arxiv_id":"2312.08722","repositories_listed":1,"syntology":null},{"url":"/paper/an-evaluation-framework-for-mapping-news","slug":"an-evaluation-framework-for-mapping-news","title":"An Evaluation Framework for Mapping News Headlines to Event Classes in a Knowledge Graph","date":"2023-12-04","arxiv_id":"2312.02334","repositories_listed":1,"syntology":null},{"url":"/paper/amrfact-enhancing-summarization-factuality","slug":"amrfact-enhancing-summarization-factuality","title":"AMRFact: Enhancing Summarization Factuality Evaluation with AMR-Driven Negative Samples Generation","date":"2023-11-16","arxiv_id":"2311.09521","repositories_listed":1,"syntology":null},{"url":"/paper/performance-trade-offs-of-watermarking-large","slug":"performance-trade-offs-of-watermarking-large","title":"Downstream Trade-offs of a Family of Text Watermarks","date":"2023-11-16","arxiv_id":"2311.09816","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/performance-trade-offs-of-watermarking-large#ran","syntology_url":"https://syntology.ai/paper/2311.09816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.09816"}},"official":{"repos":["flair-iisc/watermark_tradeoffs"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/think-while-you-write-hypothesis-verification","slug":"think-while-you-write-hypothesis-verification","title":"Think While You Write: Hypothesis Verification Promotes Faithful Knowledge-to-Text Generation","date":"2023-11-16","arxiv_id":"2311.09467","repositories_listed":1,"syntology":null},{"url":"/paper/rrescue-ranking-llm-responses-to-enhance","slug":"rrescue-ranking-llm-responses-to-enhance","title":"Rescue: Ranking LLM Responses with Partial Ordering to Improve Response Generation","date":"2023-11-15","arxiv_id":"2311.09136","repositories_listed":1,"syntology":null},{"url":"/paper/towards-label-embedding-measuring","slug":"towards-label-embedding-measuring","title":"Human-in-the-loop: Towards Label Embeddings for Measuring Classification Difficulty","date":"2023-11-15","arxiv_id":"2311.08874","repositories_listed":1,"syntology":null},{"url":"/paper/in-search-of-the-long-tail-systematic","slug":"in-search-of-the-long-tail-systematic","title":"In Search of the Long-Tail: Systematic Generation of Long-Tail Inferential Knowledge via Logical Rule Guided Search","date":"2023-11-13","arxiv_id":"2311.07237","repositories_listed":1,"syntology":null},{"url":"/paper/semi-automatic-data-enhancement-for-document","slug":"semi-automatic-data-enhancement-for-document","title":"Semi-automatic Data Enhancement for Document-Level Relation Extraction with Distant Supervision from Large Language Models","date":"2023-11-13","arxiv_id":"2311.07314","repositories_listed":1,"syntology":null},{"url":"/paper/using-natural-language-explanations-to-1","slug":"using-natural-language-explanations-to-1","title":"Using Natural Language Explanations to Improve Robustness of In-context Learning","date":"2023-11-13","arxiv_id":"2311.07556","repositories_listed":1,"syntology":null}],"record_sha256":"8c313179c35e8af02b0c6331d47c9bd4ce367e3c03a36183511c97abbee102a1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}