{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/weight-decay/papers/ran/3","list_of":"/method/weight-decay","method":"Weight Decay","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":13,"rows_per_page":100,"rows":[201,300],"of":1291,"counts":{"archive_papers_tagged":10713,"with_a_code_link":4533,"where_syntology_ran_a_sample":1291,"not_listed_spam_title":0,"listed":10713,"listed_where_code_ran":1291,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1064,"every_run_a_failure_of_syntologys_instrument":227,"listed_with_a_run_with_no_instrument_failure":1064,"listed_every_run_a_failure_of_syntologys_instrument":227,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/weight-decay/papers/ran/1","prev":"/method/weight-decay/papers/ran/2","next":"/method/weight-decay/papers/ran/4","papers":[{"paper":"/paper/model-internals-based-answer-attribution-for","slug":"model-internals-based-answer-attribution-for","title":"Model Internals-based Answer Attribution for Trustworthy Retrieval-Augmented Generation","date":"2024-06-19","arxiv_id":"2406.13663","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["betswish/mirage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-a-client-centered-assessment-of-llm","slug":"towards-a-client-centered-assessment-of-llm","title":"Towards a Client-Centered Assessment of LLM Therapists by Client Simulation","date":"2024-06-18","arxiv_id":"2406.12266","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wangjs9/clientcast"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/planrag-a-plan-then-retrieval-augmented","slug":"planrag-a-plan-then-retrieval-augmented","title":"PlanRAG: A Plan-then-Retrieval Augmented Generation for Generative Large Language Models as Decision Makers","date":"2024-06-18","arxiv_id":"2406.12430","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["myeon9h/planrag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/textit-refiner-restructure-retrieval-content","slug":"textit-refiner-restructure-retrieval-content","title":"Refiner: Restructure Retrieval Content Efficiently to Advance Question-Answering Capabilities","date":"2024-06-17","arxiv_id":"2406.11357","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allen-li1231/refiner-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/trace-the-evidence-constructing-knowledge","slug":"trace-the-evidence-constructing-knowledge","title":"TRACE the Evidence: Constructing Knowledge-Grounded Reasoning Chains for Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11460","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["jyfang6/trace"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-the-codebook-size-of-vqgan-to-100000","slug":"scaling-the-codebook-size-of-vqgan-to-100000","title":"Scaling the Codebook Size of VQGAN to 100,000 with a Utilization Rate of 99%","date":"2024-06-17","arxiv_id":"2406.11837","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zh460045050/vqgan-lc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/welldunn-on-the-robustness-and-explainability","slug":"welldunn-on-the-robustness-and-explainability","title":"WellDunn: On the Robustness and Explainability of Language Models and Large Language Models in Identifying Wellness Dimensions","date":"2024-06-17","arxiv_id":"2406.12058","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vedantpalit/WellDunn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":0,"n_instrument":5,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/fine-tuned-small-llms-still-significantly","slug":"fine-tuned-small-llms-still-significantly","title":"Fine-Tuned 'Small' LLMs (Still) Significantly Outperform Zero-Shot Generative AI Models in Text Classification","date":"2024-06-12","arxiv_id":"2406.08660","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mnbucher/text-cls-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/compute-better-spent-replacing-dense-layers","slug":"compute-better-spent-replacing-dense-layers","title":"Compute Better Spent: Replacing Dense Layers with Structured Matrices","date":"2024-06-10","arxiv_id":"2406.06248","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shikaiqiu/compute-better-spent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/domainrag-a-chinese-benchmark-for-evaluating","slug":"domainrag-a-chinese-benchmark-for-evaluating","title":"DomainRAG: A Chinese Benchmark for Evaluating Domain-specific Retrieval-Augmented Generation","date":"2024-06-09","arxiv_id":"2406.05654","n_code_links":2,"syntology":{"ran":12,"of":12,"n_ran_checked":11,"n_instrument":1,"unverified":0,"pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ShootingWong/DomainRAG"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/crag-comprehensive-rag-benchmark","slug":"crag-comprehensive-rag-benchmark","title":"CRAG -- Comprehensive RAG Benchmark","date":"2024-06-07","arxiv_id":"2406.04744","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":3,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/crag"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/berts-are-generative-in-context-learners","slug":"berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","arxiv_id":"2406.04823","n_code_links":1,"syntology":{"ran":13,"of":26,"n_ran_checked":12,"n_instrument":1,"unverified":13,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 13 unverified","official":{"repos":["ltgoslo/bert-in-context"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":13,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-subjective-uncertainty-quantification-and","slug":"on-subjective-uncertainty-quantification-and","title":"On Subjective Uncertainty Quantification and Calibration in Natural Language Generation","date":"2024-06-07","arxiv_id":"2406.05213","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["meta-inf/suq-nlg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gamebench-evaluating-strategic-reasoning","slug":"gamebench-evaluating-strategic-reasoning","title":"GameBench: Evaluating Strategic Reasoning Abilities of LLM Agents","date":"2024-06-07","arxiv_id":"2406.06613","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Joshuaclymer/GameBench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/your-absorbing-discrete-diffusion-secretly","slug":"your-absorbing-discrete-diffusion-secretly","title":"Your Absorbing Discrete Diffusion Secretly Models the Conditional Distributions of Clean Data","date":"2024-06-06","arxiv_id":"2406.03736","n_code_links":2,"syntology":{"ran":13,"of":13,"n_ran_checked":12,"n_instrument":1,"unverified":0,"pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ml-gsai/radd"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/simplified-and-generalized-masked-diffusion","slug":"simplified-and-generalized-masked-diffusion","title":"Simplified and Generalized Masked Diffusion for Discrete Data","date":"2024-06-06","arxiv_id":"2406.04329","n_code_links":1,"syntology":{"ran":6,"of":21,"n_ran_checked":2,"n_instrument":4,"unverified":15,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 15 unverified","official":{"repos":["google-deepmind/md4"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":15,"ran_from_kinds":["official"]}}},{"paper":"/paper/randomized-geometric-algebra-methods-for","slug":"randomized-geometric-algebra-methods-for","title":"Randomized Geometric Algebra Methods for Convex Neural Networks","date":"2024-06-04","arxiv_id":"2406.02806","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pilancilab/Randomized-Geometric-Algebra-Methods-for-Convex-Neural-Networks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/semcoder-training-code-language-models-with","slug":"semcoder-training-code-language-models-with","title":"SemCoder: Training Code Language Models with Comprehensive Semantics Reasoning","date":"2024-06-03","arxiv_id":"2406.01006","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":10,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["arise-lab/semcoder"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-mathematical-reasoning-of-large","slug":"evaluating-mathematical-reasoning-of-large","title":"Evaluating Mathematical Reasoning of Large Language Models: A Focus on Error Identification and Correction","date":"2024-06-02","arxiv_id":"2406.00755","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["littlecirc1e/eic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-are-zero-shot-next","slug":"large-language-models-are-zero-shot-next","title":"Large Language Models are Zero-Shot Next Location Predictors","date":"2024-05-31","arxiv_id":"2405.20962","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ssai-trento/llm-zero-shot-nl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-noise-robustness-of-retrieval","slug":"enhancing-noise-robustness-of-retrieval","title":"Enhancing Noise Robustness of Retrieval-Augmented Language Models with Adaptive Adversarial Training","date":"2024-05-31","arxiv_id":"2405.20978","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["calubkk/raat"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llamea-a-large-language-model-evolutionary","slug":"llamea-a-large-language-model-evolutionary","title":"LLaMEA: A Large Language Model Evolutionary Algorithm for Automatically Generating Metaheuristics","date":"2024-05-30","arxiv_id":"2405.20132","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nikivanstein/LLaMEA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/gnn-rag-graph-neural-retrieval-for-large","slug":"gnn-rag-graph-neural-retrieval-for-large","title":"GNN-RAG: Graph Neural Retrieval for Large Language Model Reasoning","date":"2024-05-30","arxiv_id":"2405.20139","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cmavro/gnn-rag"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/anah-analytical-annotation-of-hallucinations","slug":"anah-analytical-annotation-of-hallucinations","title":"ANAH: Analytical Annotation of Hallucinations in Large Language Models","date":"2024-05-30","arxiv_id":"2405.20315","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["open-compass/anah"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/ctrla-adaptive-retrieval-augmented-generation","slug":"ctrla-adaptive-retrieval-augmented-generation","title":"CtrlA: Adaptive Retrieval-Augmented Generation via Inherent Control","date":"2024-05-29","arxiv_id":"2405.18727","n_code_links":1,"syntology":{"ran":7,"of":12,"n_ran_checked":7,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["hsliu-initial/ctrla"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/map-neo-highly-capable-and-transparent","slug":"map-neo-highly-capable-and-transparent","title":"MAP-Neo: Highly Capable and Transparent Bilingual Large Language Model Series","date":"2024-05-29","arxiv_id":"2405.19327","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["multimodal-art-projection/map-neo"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/aligning-to-thousands-of-preferences-via","slug":"aligning-to-thousands-of-preferences-via","title":"Aligning to Thousands of Preferences via System Message Generalization","date":"2024-05-28","arxiv_id":"2405.17977","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kaistAI/Janus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"paper":"/paper/atm-adversarial-tuning-multi-agent-system","slug":"atm-adversarial-tuning-multi-agent-system","title":"ATM: Adversarial Tuning Multi-agent System Makes a Robust Retrieval-Augmented Generator","date":"2024-05-28","arxiv_id":"2405.18111","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["chuhac/atm-rag"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-empirical-analysis-on-large-language","slug":"an-empirical-analysis-on-large-language","title":"An Empirical Analysis on Large Language Models in Debate Evaluation","date":"2024-05-28","arxiv_id":"2406.00050","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["xinyiliu0227/llm_debate_bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/thread-thinking-deeper-with-recursive","slug":"thread-thinking-deeper-with-recursive","title":"THREAD: Thinking Deeper with Recursive Spawning","date":"2024-05-27","arxiv_id":"2405.17402","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["philipmit/thread"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/inversionview-a-general-purpose-method-for","slug":"inversionview-a-general-purpose-method-for","title":"InversionView: A General-Purpose Method for Reading Information from Neural Activations","date":"2024-05-27","arxiv_id":"2405.17653","n_code_links":1,"syntology":{"ran":1,"of":6,"n_ran_checked":1,"n_instrument":0,"unverified":5,"pointer_only":6,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["huangxt39/inversionview"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/accelerating-transformers-with-spectrum-1","slug":"accelerating-transformers-with-spectrum-1","title":"Accelerating Transformers with Spectrum-Preserving Token Merging","date":"2024-05-25","arxiv_id":"2405.16148","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hchautran/PiToMe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/automanual-generating-instruction-manuals-by","slug":"automanual-generating-instruction-manuals-by","title":"AutoManual: Constructing Instruction Manuals by LLM Agents via Interactive Environmental Learning","date":"2024-05-25","arxiv_id":"2405.16247","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["minghchen/automanual"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-the-language-of-protein-structure","slug":"learning-the-language-of-protein-structure","title":"Learning the Language of Protein Structure","date":"2024-05-24","arxiv_id":"2405.15840","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["instadeepai/protein-structure-tokenizer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-the-adversarial-robustness-of-1","slug":"evaluating-the-adversarial-robustness-of-1","title":"Evaluating and Safeguarding the Adversarial Robustness of Retrieval-Based In-Context Learning","date":"2024-05-24","arxiv_id":"2405.15984","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["simonucl/adv-retreival-icl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/phinets-brain-inspired-non-contrastive","slug":"phinets-brain-inspired-non-contrastive","title":"PhiNets: Brain-inspired Non-contrastive Learning Based on Temporal Prediction Hypothesis","date":"2024-05-23","arxiv_id":"2405.14650","n_code_links":0,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/wise-rethinking-the-knowledge-memory-for","slug":"wise-rethinking-the-knowledge-memory-for","title":"WISE: Rethinking the Knowledge Memory for Lifelong Model Editing of Large Language Models","date":"2024-05-23","arxiv_id":"2405.14768","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":4,"n_instrument":9,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/editworld-simulating-world-dynamics-for","slug":"editworld-simulating-world-dynamics-for","title":"EditWorld: Simulating World Dynamics for Instruction-Following Image Editing","date":"2024-05-23","arxiv_id":"2405.14785","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yangling0818/editworld"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hipporag-neurobiologically-inspired-long-term","slug":"hipporag-neurobiologically-inspired-long-term","title":"HippoRAG: Neurobiologically Inspired Long-Term Memory for Large Language Models","date":"2024-05-23","arxiv_id":"2405.14831","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["osu-nlp-group/hipporag"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/from-explicit-cot-to-implicit-cot-learning-to","slug":"from-explicit-cot-to-implicit-cot-learning-to","title":"From Explicit CoT to Implicit CoT: Learning to Internalize CoT Step by Step","date":"2024-05-23","arxiv_id":"2405.14838","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["da03/internalize_cot_step_by_step"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/not-all-language-model-features-are-linear","slug":"not-all-language-model-features-are-linear","title":"Not All Language Model Features Are Linear","date":"2024-05-23","arxiv_id":"2405.14860","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["joshengels/multidimensionalfeatures"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/ceebert-cross-domain-inference-in-early-exit","slug":"ceebert-cross-domain-inference-in-early-exit","title":"CEEBERT: Cross-Domain Inference in Early Exit BERT","date":"2024-05-23","arxiv_id":"2405.15039","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["Div290/CeeBERT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/eliciting-informative-text-evaluations-with","slug":"eliciting-informative-text-evaluations-with","title":"Eliciting Informative Text Evaluations with Large Language Models","date":"2024-05-23","arxiv_id":"2405.15077","n_code_links":1,"syntology":{"ran":10,"of":14,"n_ran_checked":8,"n_instrument":2,"unverified":4,"pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["yx-lu/eliciting-informative-text-evaluations-with-large-language-models"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/topa-extend-large-language-models-for-video","slug":"topa-extend-large-language-models-for-video","title":"TOPA: Extending Large Language Models for Video Understanding via Text-Only Pre-Alignment","date":"2024-05-22","arxiv_id":"2405.13911","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":4,"n_instrument":2,"unverified":3,"pointer_only":3,"phrase":"6 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dhg-wei/topa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/quantifying-emergence-in-large-language","slug":"quantifying-emergence-in-large-language","title":"Quantifying Semantic Emergence in Language Models","date":"2024-05-21","arxiv_id":"2405.12617","n_code_links":1,"syntology":{"ran":1,"of":6,"n_ran_checked":1,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["zodiark-ch/emergence-of-llms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-stance-detection-using-contextual","slug":"zero-shot-stance-detection-using-contextual","title":"Zero-Shot Stance Detection using Contextual Data Generation with LLMs","date":"2024-05-19","arxiv_id":"2405.11637","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Babakbehkamkia/GPT-Stance-Detection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/your-transformer-is-secretly-linear","slug":"your-transformer-is-secretly-linear","title":"Your Transformer is Secretly Linear","date":"2024-05-19","arxiv_id":"2405.12250","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AIRI-Institute/LLM-Microscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-models-can-exploit-cross-task-in","slug":"language-models-can-exploit-cross-task-in","title":"Language Models can Exploit Cross-Task In-context Learning for Data-Scarce Novel Tasks","date":"2024-05-17","arxiv_id":"2405.10548","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["c-anwoy/cross-task-icl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/freeva-offline-mllm-as-training-free-video","slug":"freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","arxiv_id":"2405.07798","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":4,"n_instrument":4,"unverified":1,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["whwu95/freeva"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/limited-ability-of-llms-to-simulate-human","slug":"limited-ability-of-llms-to-simulate-human","title":"Limited Ability of LLMs to Simulate Human Psychological Behaviours: a Psychometric Analysis","date":"2024-05-12","arxiv_id":"2405.07248","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nikbpetrov/llms-simulate-humans"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-does-gpt-2-predict-acronyms-extracting","slug":"how-does-gpt-2-predict-acronyms-extracting","title":"How does GPT-2 Predict Acronyms? Extracting and Understanding a Circuit via Mechanistic Interpretability","date":"2024-05-07","arxiv_id":"2405.04156","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":1,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jgcarrasco/acronyms_paper"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/revisiting-character-level-adversarial","slug":"revisiting-character-level-adversarial","title":"Revisiting Character-level Adversarial Attacks for Language Models","date":"2024-05-07","arxiv_id":"2405.04346","n_code_links":1,"syntology":{"ran":23,"of":31,"n_ran_checked":13,"n_instrument":10,"unverified":8,"pointer_only":0,"phrase":"23 ran (of which 3 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 10 where Syntology's instrument failed) · 8 unverified","official":{"repos":["lions-epfl/charmer"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":3,"n_ran_no_instrument_failure":13,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/hire-me-or-not-examining-language-model-s","slug":"hire-me-or-not-examining-language-model-s","title":"Hire Me or Not? Examining Language Model's Behavior with Occupation Attributes","date":"2024-05-06","arxiv_id":"2405.06687","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["daminz97/multi-step_gsv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/do-large-language-models-understand","slug":"do-large-language-models-understand","title":"Do Large Language Models Understand Conversational Implicature -- A case study with a chinese sitcom","date":"2024-04-30","arxiv_id":"2404.19509","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":6,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["sjtu-compling/llm-pragmatics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/worldvaluesbench-a-large-scale-benchmark","slug":"worldvaluesbench-a-large-scale-benchmark","title":"WorldValuesBench: A Large-Scale Benchmark Dataset for Multi-Cultural Value Awareness of Language Models","date":"2024-04-25","arxiv_id":"2404.16308","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["demon702/worldvaluesbench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/studying-large-language-model-behaviors-under","slug":"studying-large-language-model-behaviors-under","title":"Studying Large Language Model Behaviors Under Context-Memory Conflicts With Real Documents","date":"2024-04-24","arxiv_id":"2404.16032","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kortukov/realistic_knowledge_conflicts"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/from-local-to-global-a-graph-rag-approach-to","slug":"from-local-to-global-a-graph-rag-approach-to","title":"From Local to Global: A Graph RAG Approach to Query-Focused Summarization","date":"2024-04-24","arxiv_id":"2404.16130","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/typos-that-broke-the-rag-s-back-genetic","slug":"typos-that-broke-the-rag-s-back-genetic","title":"Typos that Broke the RAG's Back: Genetic Attack on RAG Pipeline by Simulating Documents in the Wild via Low-level Perturbations","date":"2024-04-22","arxiv_id":"2404.13948","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["zomss/garag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-well-can-llms-echo-us-evaluating-ai","slug":"how-well-can-llms-echo-us-evaluating-ai","title":"How Well Can LLMs Echo Us? Evaluating AI Chatbots' Role-Play Ability with ECHO","date":"2024-04-22","arxiv_id":"2404.13957","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":0,"n_instrument":6,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cuhk-arise/echo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/calc-cmu-at-semeval-2024-task-7-pre-calc","slug":"calc-cmu-at-semeval-2024-task-7-pre-calc","title":"Pre-Calc: Learning to Use the Calculator Improves Numeracy in Language Models","date":"2024-04-22","arxiv_id":"2404.14355","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["calc-cmu/pre-calc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/svgeditbench-a-benchmark-dataset-for","slug":"svgeditbench-a-benchmark-dataset-for","title":"SVGEditBench: A Benchmark Dataset for Quantitative Assessment of LLM's SVG Editing Capabilities","date":"2024-04-21","arxiv_id":"2404.13710","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mti-lab/svgeditbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-retrieval-quality-in-retrieval","slug":"evaluating-retrieval-quality-in-retrieval","title":"Evaluating Retrieval Quality in Retrieval-Augmented Generation","date":"2024-04-21","arxiv_id":"2404.13781","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alirezasalemi7/erag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/longembed-extending-embedding-models-for-long","slug":"longembed-extending-embedding-models-for-long","title":"LongEmbed: Extending Embedding Models for Long Context Retrieval","date":"2024-04-18","arxiv_id":"2404.12096","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":2,"n_instrument":3,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dwzhu-pku/longembed"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/from-form-s-to-meaning-probing-the-semantic","slug":"from-form-s-to-meaning-probing-the-semantic","title":"From Form(s) to Meaning: Probing the Semantic Depths of Language Models Using Multisense Consistency","date":"2024-04-18","arxiv_id":"2404.12145","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["facebookresearch/multisense_consistency"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/decoupled-weight-decay-for-any-p-norm","slug":"decoupled-weight-decay-for-any-p-norm","title":"Decoupled Weight Decay for Any $p$ Norm","date":"2024-04-16","arxiv_id":"2404.10824","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":9,"n_instrument":0,"unverified":2,"pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["nadav-out/padam"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/pre-training-small-base-lms-with-fewer-tokens","slug":"pre-training-small-base-lms-with-fewer-tokens","title":"Inheritune: Training Smaller Yet More Attentive Language Models","date":"2024-04-12","arxiv_id":"2404.08634","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sanyalsunny111/llm-inheritune"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-training-data-influence-of-gpt-models","slug":"on-training-data-influence-of-gpt-models","title":"On Training Data Influence of GPT Models","date":"2024-04-11","arxiv_id":"2404.07840","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["eleutherai/pythia","ernie-research/gptfluence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-the-potential-of-large-foundation","slug":"exploring-the-potential-of-large-foundation","title":"Exploring the Potential of Large Foundation Models for Open-Vocabulary HOI Detection","date":"2024-04-09","arxiv_id":"2404.06194","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ltttpku/cmd-se-release"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/banglaautokg-automatic-bangla-knowledge-graph","slug":"banglaautokg-automatic-bangla-knowledge-graph","title":"BanglaAutoKG: Automatic Bangla Knowledge Graph Construction with Semantic Neural Graph Filtering","date":"2024-04-04","arxiv_id":"2404.03528","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["azminewasi/banglaautokg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/outlier-efficient-hopfield-layers-for-large","slug":"outlier-efficient-hopfield-layers-for-large","title":"Outlier-Efficient Hopfield Layers for Large Transformer-Based Models","date":"2024-04-04","arxiv_id":"2404.03828","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":5,"n_instrument":5,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","official":{"repos":["magics-lab/outeffhop"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/utebc-nlp-at-semeval-2024-task-9-can-llms-be","slug":"utebc-nlp-at-semeval-2024-task-9-can-llms-be","title":"uTeBC-NLP at SemEval-2024 Task 9: Can LLMs be Lateral Thinkers?","date":"2024-04-03","arxiv_id":"2404.02474","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ipouyall/can-llms-be-lateral-thinkers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scene-adaptive-sparse-transformer-for-event","slug":"scene-adaptive-sparse-transformer-for-event","title":"Scene Adaptive Sparse Transformer for Event-based Object Detection","date":"2024-04-02","arxiv_id":"2404.01882","n_code_links":1,"syntology":{"ran":19,"of":23,"n_ran_checked":19,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["peterande/sast"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":19,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/advancing-llm-reasoning-generalists-with","slug":"advancing-llm-reasoning-generalists-with","title":"Advancing LLM Reasoning Generalists with Preference Trees","date":"2024-04-02","arxiv_id":"2404.02078","n_code_links":1,"syntology":{"ran":18,"of":20,"n_ran_checked":13,"n_instrument":5,"unverified":2,"pointer_only":2,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 1 violated, 12 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["openbmb/eurus"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/ginopic-topic-modeling-with-graph-isomorphism","slug":"ginopic-topic-modeling-with-graph-isomorphism","title":"GINopic: Topic Modeling with Graph Isomorphism Network","date":"2024-04-02","arxiv_id":"2404.02115","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":9,"n_instrument":1,"unverified":2,"pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["adhyasuman/ginopic"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/jailbreaking-leading-safety-aligned-llms-with","slug":"jailbreaking-leading-safety-aligned-llms-with","title":"Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks","date":"2024-04-02","arxiv_id":"2404.02151","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":7,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tml-epfl/llm-adaptive-attacks"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fables-evaluating-faithfulness-and-content","slug":"fables-evaluating-faithfulness-and-content","title":"FABLES: Evaluating faithfulness and content selection in book-length summarization","date":"2024-04-01","arxiv_id":"2404.01261","n_code_links":3,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mungg/fables"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/are-large-language-models-good-at-utility","slug":"are-large-language-models-good-at-utility","title":"Are Large Language Models Good at Utility Judgments?","date":"2024-03-28","arxiv_id":"2403.19216","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ict-bigdatalab/utility_judgments"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpreting-key-mechanisms-of-factual-recall","slug":"interpreting-key-mechanisms-of-factual-recall","title":"Interpreting Key Mechanisms of Factual Recall in Transformer-Based Language Models","date":"2024-03-28","arxiv_id":"2403.19521","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["trestad/factual-recall-mechanism"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/semrode-macro-adversarial-training-to-learn","slug":"semrode-macro-adversarial-training-to-learn","title":"SemRoDe: Macro Adversarial Training to Learn Representations That are Robust to Word-Level Attacks","date":"2024-03-27","arxiv_id":"2403.18423","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aniloid2/semrode-macroadversarialtraining"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/vulnerability-detection-with-code-language","slug":"vulnerability-detection-with-code-language","title":"Vulnerability Detection with Code Language Models: How Far Are We?","date":"2024-03-27","arxiv_id":"2403.18624","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["dlvuldet/primevul"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/long-form-factuality-in-large-language-models","slug":"long-form-factuality-in-large-language-models","title":"Long-form factuality in large language models","date":"2024-03-27","arxiv_id":"2403.18802","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-deepmind/long-form-factuality"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/omnivid-a-generative-framework-for-universal","slug":"omnivid-a-generative-framework-for-universal","title":"OmniVid: A Generative Framework for Universal Video Understanding","date":"2024-03-26","arxiv_id":"2403.17935","n_code_links":1,"syntology":{"ran":7,"of":11,"n_ran_checked":6,"n_instrument":1,"unverified":4,"pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["wangjk666/omnivid"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/don-t-trust-verify-grounding-llm-quantitative","slug":"don-t-trust-verify-grounding-llm-quantitative","title":"Don't Trust: Verify -- Grounding LLM Quantitative Reasoning with Autoformalization","date":"2024-03-26","arxiv_id":"2403.18120","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":2,"n_instrument":8,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jinpz/dtv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/repairagent-an-autonomous-llm-based-agent-for","slug":"repairagent-an-autonomous-llm-based-agent-for","title":"RepairAgent: An Autonomous, LLM-Based Agent for Program Repair","date":"2024-03-25","arxiv_id":"2403.17134","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sola-st/RepairAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/blended-rag-improving-rag-retriever-augmented","slug":"blended-rag-improving-rag-retriever-augmented","title":"Blended RAG: Improving RAG (Retriever-Augmented Generation) Accuracy with Semantic Search and Hybrid Query-Based Retrievers","date":"2024-03-22","arxiv_id":"2404.07220","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ibm-ecosystem-engineering/blended-rag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/psalm-pixelwise-segmentation-with-large-multi","slug":"psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","arxiv_id":"2403.14598","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zamling/psalm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/emergent-world-models-and-latent-variable","slug":"emergent-world-models-and-latent-variable","title":"Emergent World Models and Latent Variable Estimation in Chess-Playing Language Models","date":"2024-03-21","arxiv_id":"2403.15498","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["adamkarvonen/chess_llm_interpretability"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/embedded-named-entity-recognition-using","slug":"embedded-named-entity-recognition-using","title":"Embedded Named Entity Recognition using Probing Classifiers","date":"2024-03-18","arxiv_id":"2403.11747","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":4,"n_instrument":0,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nicpopovic/stoke","nicpopovic/ember"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/meta-prompting-for-automating-zero-shot","slug":"meta-prompting-for-automating-zero-shot","title":"Meta-Prompting for Automating Zero-shot Visual Recognition with LLMs","date":"2024-03-18","arxiv_id":"2403.11755","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jmiemirza/meta-prompting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-far-are-we-on-the-decision-making-of-llms","slug":"how-far-are-we-on-the-decision-making-of-llms","title":"How Far Are We on the Decision-Making of LLMs? Evaluating LLMs' Gaming Ability in Multi-Agent Environments","date":"2024-03-18","arxiv_id":"2403.11807","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cuhk-arise/gamabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/easyjailbreak-a-unified-framework-for","slug":"easyjailbreak-a-unified-framework-for","title":"EasyJailbreak: A Unified Framework for Jailbreaking Large Language Models","date":"2024-03-18","arxiv_id":"2403.12171","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["easyjailbreak/easyjailbreak"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/data-is-all-you-need-finetuning-llms-for-chip","slug":"data-is-all-you-need-finetuning-llms-for-chip","title":"Data is all you need: Finetuning LLMs for Chip Design via an Automated design-data augmentation framework","date":"2024-03-17","arxiv_id":"2403.11202","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aichipdesign/chipgptft"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dragin-dynamic-retrieval-augmented-generation","slug":"dragin-dynamic-retrieval-augmented-generation","title":"DRAGIN: Dynamic Retrieval Augmented Generation based on the Information Needs of Large Language Models","date":"2024-03-15","arxiv_id":"2403.10081","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["oneal2000/dragin"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-llm-factual-accuracy-with-rag-to","slug":"enhancing-llm-factual-accuracy-with-rag-to","title":"Enhancing LLM Factual Accuracy with RAG to Counter Hallucinations: A Case Study on Domain-Specific Queries in Private Knowledge-Bases","date":"2024-03-15","arxiv_id":"2403.10446","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":12,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["anlp-team/LTI_Neural_Navigator"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/codeultrafeedback-an-llm-as-a-judge-dataset","slug":"codeultrafeedback-an-llm-as-a-judge-dataset","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","date":"2024-03-14","arxiv_id":"2403.09032","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["martin-wey/codeultrafeedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ragged-towards-informed-design-of-retrieval","slug":"ragged-towards-informed-design-of-retrieval","title":"RAGGED: Towards Informed Design of Retrieval Augmented Generation Systems","date":"2024-03-14","arxiv_id":"2403.09040","n_code_links":1,"syntology":{"ran":13,"of":14,"n_ran_checked":12,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neulab/ragged"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rectifying-demonstration-shortcut-in-in","slug":"rectifying-demonstration-shortcut-in-in","title":"Rectifying Demonstration Shortcut in In-Context Learning","date":"2024-03-14","arxiv_id":"2403.09488","n_code_links":1,"syntology":{"ran":3,"of":9,"n_ran_checked":1,"n_instrument":2,"unverified":6,"pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["lainshower/in-context-calibration"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/optimistic-verifiable-training-by-controlling","slug":"optimistic-verifiable-training-by-controlling","title":"Optimistic Verifiable Training by Controlling Hardware Nondeterminism","date":"2024-03-14","arxiv_id":"2403.09603","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":4,"n_instrument":5,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["meghabyte/verifiable-training"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/generative-pretrained-structured-transformers","slug":"generative-pretrained-structured-transformers","title":"Generative Pretrained Structured Transformers: Unsupervised Syntactic Language Models at Scale","date":"2024-03-13","arxiv_id":"2403.08293","n_code_links":2,"syntology":{"ran":4,"of":7,"n_ran_checked":3,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ant-research/structuredlm_rtdt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"76bed220ba5ea8e4d45101b36acd7ea97ac08f93248af5baccd13949bfebceb7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}